From d24240c014ead42747f7830cc366e26a273cd371 Mon Sep 17 00:00:00 2001 From: ishaan-berri <155045088+ishaan-berri@users.noreply.github.com> Date: Thu, 1 Oct 2026 10:55:51 -0700 Subject: [PATCH] feat(ui): agent traces open in a side drawer with a chat-style run view (#43972) * feat(ui): redesign agent trace run view with chat-style detail pane Tree with connector lines, typed icon tiles and provider logos, hover cards with timing, and Input/Output sections rendered as message cards. * fix(tracing): show text for block-list message content and split normalizers per convention OpenAI responses-style content (reasoning + text blocks) rendered as raw JSON in the trace view. Keep the text blocks and drop opaque reasoning. Move each convention into litellm/tracing/normalizers with an ordered registry so new frameworks plug in without touching OTLP decoding. * fix(tracing): keep long message histories as valid JSON and parse function_call blocks * feat(ui): open agent traces in a resizable side drawer with a devtool-style tree Clicking a run opens it in a drawer over the list instead of a full page. j/k and the header arrows switch runs, Esc closes. The tree gets dashed connectors, per-span waterfall bars, mono tool names and real provider logos. AI messages with reasoning/function_call blocks render as text. * fix(trace-ui): address review: valid JSON trimming, drawer keys, reduced motion, narrow screens * feat(tracing): serve span content in a standard LiteLLM UI format GET /v1/traces/{trace_id}/spans/{span_id} now also returns input_ui and output_ui, a tagged union of messages, fields or text built server side by litellm/tracing/ui_format.py. The trace UI renders from those fields and only falls back to client-side parsing when talking to an older proxy. The raw input and output strings are unchanged, and so is storage Co-Authored-By: Claude Opus 5.5 * fix(tracing): fall back to an elision marker when shortened messages still exceed the size limit * fix(tracing): keep both messages when tool_calls are oversized and keep failed-tool styling --------- Co-authored-by: Claude Opus 5.5 --- litellm/tracing/decode.py | 252 +++------ litellm/tracing/normalizers/__init__.py | 32 ++ litellm/tracing/normalizers/base.py | 22 + litellm/tracing/normalizers/genai.py | 31 + litellm/tracing/normalizers/langsmith.py | 115 ++++ litellm/tracing/normalizers/messages.py | 59 ++ litellm/tracing/normalizers/openinference.py | 27 + litellm/tracing/store.py | 3 + litellm/tracing/types.py | 4 + litellm/tracing/ui_format.py | 158 ++++++ .../proxy/test_tracing_endpoints.py | 21 +- .../tracing/normalizers/test_registry.py | 68 +++ tests/test_litellm/tracing/test_decode.py | 101 ++++ tests/test_litellm/tracing/test_store.py | 11 +- tests/test_litellm/tracing/test_ui_format.py | 119 ++++ ui/litellm-dashboard/src/app/globals.css | 167 ++++++ .../TraceView/AgentTracesSection.test.tsx | 44 +- .../TraceView/AgentTracesSection.tsx | 32 +- .../view_logs/TraceView/AgentTracesTable.tsx | 11 +- .../view_logs/TraceView/AttributesDetail.tsx | 37 +- .../view_logs/TraceView/Collapse.tsx | 35 ++ .../view_logs/TraceView/DetailContent.tsx | 203 ++++--- .../view_logs/TraceView/DetailPane.test.tsx | 145 ++++- .../view_logs/TraceView/DetailPane.tsx | 131 +++-- .../view_logs/TraceView/DurationBar.tsx | 27 - .../components/view_logs/TraceView/IdChip.tsx | 21 + .../view_logs/TraceView/KeyValueRows.tsx | 107 ++++ .../view_logs/TraceView/MessageCard.test.tsx | 80 +++ .../view_logs/TraceView/MessageCard.tsx | 265 +++++++++ .../view_logs/TraceView/RequestDetail.tsx | 84 ++- .../view_logs/TraceView/RunDrawer.test.tsx | 76 +++ .../view_logs/TraceView/RunDrawer.tsx | 229 ++++++++ .../view_logs/TraceView/SpanHoverCard.tsx | 185 ++++++ .../view_logs/TraceView/SpanIcon.tsx | 59 ++ .../view_logs/TraceView/SpanTree.tsx | 535 +++++++++++------- .../view_logs/TraceView/TraceDrawer.test.tsx | 16 + .../view_logs/TraceView/TraceDrawer.tsx | 120 ++-- .../view_logs/TraceView/spanProvider.test.ts | 35 ++ .../view_logs/TraceView/spanProvider.ts | 34 ++ .../view_logs/TraceView/traceTypes.ts | 31 +- .../view_logs/TraceView/traceUtils.test.ts | 41 ++ .../view_logs/TraceView/traceUtils.ts | 74 ++- 42 files changed, 3185 insertions(+), 662 deletions(-) create mode 100644 litellm/tracing/normalizers/__init__.py create mode 100644 litellm/tracing/normalizers/base.py create mode 100644 litellm/tracing/normalizers/genai.py create mode 100644 litellm/tracing/normalizers/langsmith.py create mode 100644 litellm/tracing/normalizers/messages.py create mode 100644 litellm/tracing/normalizers/openinference.py create mode 100644 litellm/tracing/ui_format.py create mode 100644 tests/test_litellm/tracing/normalizers/test_registry.py create mode 100644 tests/test_litellm/tracing/test_ui_format.py create mode 100644 ui/litellm-dashboard/src/components/view_logs/TraceView/Collapse.tsx delete mode 100644 ui/litellm-dashboard/src/components/view_logs/TraceView/DurationBar.tsx create mode 100644 ui/litellm-dashboard/src/components/view_logs/TraceView/IdChip.tsx create mode 100644 ui/litellm-dashboard/src/components/view_logs/TraceView/KeyValueRows.tsx create mode 100644 ui/litellm-dashboard/src/components/view_logs/TraceView/MessageCard.test.tsx create mode 100644 ui/litellm-dashboard/src/components/view_logs/TraceView/MessageCard.tsx create mode 100644 ui/litellm-dashboard/src/components/view_logs/TraceView/RunDrawer.test.tsx create mode 100644 ui/litellm-dashboard/src/components/view_logs/TraceView/RunDrawer.tsx create mode 100644 ui/litellm-dashboard/src/components/view_logs/TraceView/SpanHoverCard.tsx create mode 100644 ui/litellm-dashboard/src/components/view_logs/TraceView/SpanIcon.tsx create mode 100644 ui/litellm-dashboard/src/components/view_logs/TraceView/spanProvider.test.ts create mode 100644 ui/litellm-dashboard/src/components/view_logs/TraceView/spanProvider.ts diff --git a/litellm/tracing/decode.py b/litellm/tracing/decode.py index c310a339593..b1bd951220c 100644 --- a/litellm/tracing/decode.py +++ b/litellm/tracing/decode.py @@ -9,14 +9,23 @@ Pure functions, no I/O. Two steps: """ import json -from collections.abc import Callable, Mapping +from collections.abc import Mapping +from itertools import accumulate from types import MappingProxyType -from typing import Any, Final +from typing import Final + +from pydantic import JsonValue, TypeAdapter, ValidationError +from typing_extensions import ReadOnly, TypedDict from litellm.constants import OTLP_MAX_ATTRIBUTE_VALUE_BYTES, OTLP_MAX_BODY_BYTES from litellm.rust_bridge.traces import DecodedSpan from litellm.rust_bridge.traces import decode_otlp as native_decode_otlp -from litellm.tracing.types import SpanRow, SpanType +from litellm.tracing.normalizers import select_normalizer +from litellm.tracing.normalizers.base import to_int +from litellm.tracing.types import SpanRow + +_MESSAGE_LIST: Final = TypeAdapter(tuple[dict[str, JsonValue], ...]) +_MAX_JSON_ESCAPE_BYTES: Final = 6 # attributes whose content we lift into Input/Output and drop from SpanAttributes _HEAVY_ATTRIBUTES: Final = frozenset( @@ -30,18 +39,6 @@ _HEAVY_ATTRIBUTES: Final = frozenset( "output.value", } ) -# LangChain / Deep Agents middleware wrappers: real spans, but noise in the UI -_FRAMEWORK_SUFFIXES: Final = ( - ".wrap_model_call", - ".wrap_tool_call", - ".before_agent", - ".after_agent", - ".before_model", - ".after_model", -) -_LLM_OPERATIONS: Final = frozenset({"chat", "text_completion", "generate_content"}) -_LC_ROLES: Final = MappingProxyType({"human": "user", "ai": "assistant", "system": "system", "tool": "tool"}) -_OPENINFERENCE_TYPES: Final[Mapping[str, SpanType]] = MappingProxyType({"AGENT": "agent", "LLM": "llm", "TOOL": "tool"}) class InvalidOTLPPayloadError(ValueError): @@ -60,6 +57,83 @@ def _truncate(value: str) -> str: return f"{kept}…[truncated {size - OTLP_MAX_ATTRIBUTE_VALUE_BYTES} bytes]" +def _size(value: str) -> int: + return len(value.encode("utf-8")) + + +class _ElisionMarker(TypedDict): + role: ReadOnly[str] + content: ReadOnly[str] + + +def _elided(count: int) -> str: + marker: Final[_ElisionMarker] = {"role": "system", "content": f"…[{count} earlier messages truncated]"} + return json.dumps(marker) + + +def _with_content(message: Mapping[str, JsonValue], content: str) -> str: + return json.dumps(MappingProxyType({**message, "content": content}), default=lambda proxy: proxy.copy()) + + +def _shrunk_message(message: Mapping[str, JsonValue], budget: int) -> str: + """One message cut to `budget` bytes, as valid JSON. + + Shortens `content` first; if other fields (e.g. huge tool_calls) still don't fit, keeps only role + content. + """ + content: Final = message.get("content") + text: Final = content if isinstance(content, str) else json.dumps(content) + role_only: Final = MappingProxyType({"role": message.get("role", "user")}) + attempts: Final = ( + _cut_content(message, text, budget, 1), + _cut_content(role_only, text, budget, 1), + _cut_content(role_only, text, budget, _MAX_JSON_ESCAPE_BYTES), + ) + return next((attempt for attempt in attempts if _size(attempt) <= budget), attempts[-1]) + + +def _cut_content(message: Mapping[str, JsonValue], text: str, budget: int, escape_factor: int) -> str: + overhead: Final = _size(_with_content(message, "")) + room: Final = max(0, budget - overhead - 48) // escape_factor + kept: Final = text.encode("utf-8")[:room].decode("utf-8", "ignore") + return _with_content(message, f"{kept}…[truncated {_size(text) - _size(kept)} bytes]") + + +def _newest_that_fit(encoded: tuple[str, ...], budget: int) -> int: + """How many trailing messages fit in `budget` bytes (comma separators included), scanning newest first.""" + sizes: Final = tuple(_size(m) + 1 for m in reversed(encoded)) + totals: Final = tuple(accumulate(sizes)) + return next((count for count, total in enumerate(totals) if total > budget), len(totals)) + + +def _truncate_payload(value: str) -> str: + """Message arrays keep the first message, an elision marker and the newest messages that fit. + + The result is always valid JSON: if even those don't fit, the first and last messages are shortened. + Anything that isn't a message array is byte-truncated as before. + """ + if _size(value) <= OTLP_MAX_ATTRIBUTE_VALUE_BYTES or not value.startswith("["): + return _truncate(value) + try: + messages: Final = _MESSAGE_LIST.validate_json(value) + except ValidationError: + return _truncate(value) + if len(messages) < 2: + return _truncate(value) + encoded: Final = tuple(json.dumps(m) for m in messages) + marker_budget: Final = _size(_elided(len(messages))) + 1 + budget: Final = OTLP_MAX_ATTRIBUTE_VALUE_BYTES - 2 - _size(encoded[0]) - 1 - marker_budget + kept: Final = min(_newest_that_fit(encoded[1:], budget), len(messages) - 2) + if kept > 0: + tail: Final = encoded[len(encoded) - kept :] + return "[" + ", ".join((encoded[0], _elided(len(messages) - 1 - kept), *tail)) + "]" + half: Final = (OTLP_MAX_ATTRIBUTE_VALUE_BYTES - marker_budget - 4) // 2 + middle: Final = (_elided(len(messages) - 2),) if len(messages) > 2 else () + shrunk: Final = ( + "[" + ", ".join((_shrunk_message(messages[0], half), *middle, _shrunk_message(messages[-1], half))) + "]" + ) + return shrunk if _size(shrunk) <= OTLP_MAX_ATTRIBUTE_VALUE_BYTES else "[" + _elided(len(messages)) + "]" + + def decode_otlp( body: bytes, content_type: str | None = None, content_encoding: str | None = None ) -> tuple[SpanRow, ...]: @@ -116,157 +190,17 @@ def _span_row(span: DecodedSpan) -> SpanRow: row["SpanAttributes"] = { # mutable-ok: the Rust JSON bridge requires a plain dict for span attributes k: _truncate(v) for k, v in attributes.items() if k not in _HEAVY_ATTRIBUTES } - row["Input"], row["Output"] = _truncate(row["Input"]), _truncate(row["Output"]) + row["Input"], row["Output"] = _truncate_payload(row["Input"]), _truncate(row["Output"]) return row -def _loads(value: str) -> object: - try: - return json.loads(value) - except (ValueError, TypeError): - return None - - -def _lc_message(message: Mapping[str, Any]) -> dict[str, Any]: - """LangChain serialized message (or plain {role, content}) -> {role, content, tool_calls?}.""" - kwargs = message.get("kwargs", message) - role = _LC_ROLES.get(kwargs.get("type") or kwargs.get("role"), kwargs.get("role") or kwargs.get("type") or "") - content = kwargs.get("content", "") - out: dict[str, Any] = { # mutable-ok: the framework message is built for JSON serialization - "role": role, - "content": content if isinstance(content, str) else json.dumps(content), - } - if kwargs.get("tool_calls"): - out["tool_calls"] = tuple( - {"name": t.get("name"), "args": t.get("args")} # mutable-ok: JSON tool calls need object payloads - for t in kwargs["tool_calls"] - ) - if role == "tool" and kwargs.get("name"): - out["name"] = kwargs["name"] - return out - - -def _langsmith_type(row: SpanRow, attributes: Mapping[str, str]) -> SpanType: - kind = attributes.get("langsmith.span.kind", "chain") - name = row["SpanName"] - if not row["ParentSpanId"] or name == attributes.get("langsmith.metadata.lc_agent_name"): - return "agent" - if kind in ("llm", "tool"): - return kind - if name.endswith(_FRAMEWORK_SUFFIXES): - return "framework" - return "chain" - - -def _langsmith_io(row: SpanRow, attributes: Mapping[str, str]) -> None: - prompt = _loads(attributes.get("gen_ai.prompt", "")) - completion = _loads(attributes.get("gen_ai.completion", "")) - prompt_payload = prompt if isinstance(prompt, dict) else MappingProxyType({}) - if row["ObservationType"] == "llm" and isinstance(completion, dict): - messages = prompt_payload.get("messages") or ((),) - batch = messages[0] if messages and isinstance(messages[0], list) else messages - row["Input"] = ( - json.dumps(tuple(_lc_message(m) for m in batch if isinstance(m, dict))) - if isinstance(batch, (list, tuple)) - else "" - ) - generations: Final = completion.get("generations") - first: Final = generations[0] if isinstance(generations, list) and generations else None - item: Final = first[0] if isinstance(first, list) and first else None - message: Final = item.get("message") if isinstance(item, dict) else None - generation: Final = message.get("kwargs") if isinstance(message, dict) else None - if isinstance(generation, dict): - row["Output"] = json.dumps(_lc_message(generation)) - metadata: Final = generation.get("response_metadata") - row["LiteLLMRequestId"] = metadata.get("id", "") if isinstance(metadata, dict) else "" - else: - row["Output"] = attributes.get("gen_ai.completion", "") - return - if row["ObservationType"] == "tool": - output = completion.get("output", completion) if isinstance(completion, dict) else completion - if isinstance(output, dict) and "update" in output: # LangGraph Command, e.g. Deep Agents `task` - update: Final = output.get("update") - update_messages = update.get("messages") or () if isinstance(update, dict) else () - output = update_messages[-1] if update_messages else output - if isinstance(output, dict): - output = output.get("content", output) - row["Input"] = attributes.get("gen_ai.prompt", "") - row["Output"] = output if isinstance(output, str) else json.dumps(output) - return - if row["ObservationType"] == "agent": - input_messages = prompt.get("messages") if isinstance(prompt, dict) else None - output_messages = completion.get("messages") if isinstance(completion, dict) else None - # agents built with @traceable take arbitrary args, not a message list: keep the raw payload then - row["Input"] = ( - json.dumps(tuple(_lc_message(m) for m in input_messages if isinstance(m, dict))) - if input_messages - else attributes.get("gen_ai.prompt", "") - ) - row["Output"] = ( - json.dumps(_lc_message(output_messages[-1])) - if output_messages and isinstance(output_messages[-1], dict) - else attributes.get("gen_ai.completion", "") - ) - return - row["Input"] = attributes.get("gen_ai.prompt", "") - row["Output"] = attributes.get("gen_ai.completion", "") - - -def normalize_langsmith(row: SpanRow, attributes: Mapping[str, str]) -> None: - row["ObservationType"] = _langsmith_type(row, attributes) - row["AgentName"] = attributes.get("langsmith.metadata.lc_agent_name", "") - row["Model"] = attributes.get("gen_ai.request.model", "") - _langsmith_io(row, attributes) - - -def normalize_genai(row: SpanRow, attributes: Mapping[str, str]) -> None: - operation = attributes.get("gen_ai.operation.name", "") - if operation == "invoke_agent" or not row["ParentSpanId"]: - row["ObservationType"] = "agent" - elif operation in _LLM_OPERATIONS: - row["ObservationType"] = "llm" - elif operation == "execute_tool": - row["ObservationType"] = "tool" - row["AgentName"] = attributes.get("gen_ai.agent.name", "") - row["Model"] = attributes.get("gen_ai.request.model") or attributes.get("gen_ai.response.model", "") - row["LiteLLMRequestId"] = attributes.get("gen_ai.response.id", "") - row["Input"] = attributes.get("gen_ai.input.messages") or attributes.get("gen_ai.tool.call.arguments", "") - row["Output"] = attributes.get("gen_ai.output.messages") or attributes.get("gen_ai.tool.call.result", "") - - -def normalize_openinference(row: SpanRow, attributes: Mapping[str, str]) -> None: - kind = attributes.get("openinference.span.kind", "").upper() - row["ObservationType"] = _OPENINFERENCE_TYPES.get(kind, "agent" if not row["ParentSpanId"] else "chain") - row["AgentName"] = attributes.get("agent.name", "") - row["Model"] = attributes.get("llm.model_name", "") - row["Input"] = attributes.get("input.value", "") - row["Output"] = attributes.get("output.value", "") - row["InputTokens"] = _to_int(attributes.get("llm.token_count.prompt")) - row["OutputTokens"] = _to_int(attributes.get("llm.token_count.completion")) - - def _set_tokens(row: SpanRow, attributes: Mapping[str, str]) -> None: - row["InputTokens"] = _to_int(attributes.get("gen_ai.usage.input_tokens")) - row["OutputTokens"] = _to_int(attributes.get("gen_ai.usage.output_tokens")) - - -def _to_int(value: str | None) -> int: - try: - return int(value) if value else 0 - except ValueError: - return 0 - - -def select_normalizer(scope_name: str, attributes: Mapping[str, str]) -> Callable[[SpanRow, Mapping[str, str]], None]: - if scope_name == "langsmith" or "langsmith.span.kind" in attributes: - return normalize_langsmith - if "openinference.span.kind" in attributes: - return normalize_openinference - return normalize_genai + row["InputTokens"] = to_int(attributes.get("gen_ai.usage.input_tokens")) + row["OutputTokens"] = to_int(attributes.get("gen_ai.usage.output_tokens")) def normalize(row: SpanRow, attributes: Mapping[str, str]) -> None: - select_normalizer(row["ScopeName"], attributes)(row, attributes) + select_normalizer(row["ScopeName"], attributes).normalize(row, attributes) if not row["InputTokens"] and not row["OutputTokens"]: _set_tokens(row, attributes) diff --git a/litellm/tracing/normalizers/__init__.py b/litellm/tracing/normalizers/__init__.py new file mode 100644 index 00000000000..2f861330a36 --- /dev/null +++ b/litellm/tracing/normalizers/__init__.py @@ -0,0 +1,32 @@ +"""Per-convention span normalizers, tried in order: the first whose `matches()` is true wins.""" + +from collections.abc import Mapping, Sequence +from typing import Final + +from litellm.tracing.normalizers.base import SpanNormalizer +from litellm.tracing.normalizers.genai import GenAISemconvNormalizer +from litellm.tracing.normalizers.langsmith import LangSmithNormalizer +from litellm.tracing.normalizers.openinference import OpenInferenceNormalizer + +NORMALIZERS: Final[tuple[SpanNormalizer, ...]] = ( + LangSmithNormalizer(), + OpenInferenceNormalizer(), + GenAISemconvNormalizer(), +) +_FALLBACK: Final[SpanNormalizer] = GenAISemconvNormalizer() + + +def select_normalizer( + scope_name: str, attributes: Mapping[str, str], registry: Sequence[SpanNormalizer] = NORMALIZERS +) -> SpanNormalizer: + return next((n for n in registry if n.matches(scope_name, attributes)), _FALLBACK) + + +__all__ = ( + "NORMALIZERS", + "GenAISemconvNormalizer", + "LangSmithNormalizer", + "OpenInferenceNormalizer", + "SpanNormalizer", + "select_normalizer", +) diff --git a/litellm/tracing/normalizers/base.py b/litellm/tracing/normalizers/base.py new file mode 100644 index 00000000000..37735113ce2 --- /dev/null +++ b/litellm/tracing/normalizers/base.py @@ -0,0 +1,22 @@ +from collections.abc import Mapping +from typing import Protocol + +from litellm.tracing.types import SpanRow + + +class SpanNormalizer(Protocol): + """Maps one tracing convention's span attributes onto the LiteLLM `SpanRow` columns.""" + + @property + def name(self) -> str: ... + + def matches(self, scope_name: str, attributes: Mapping[str, str]) -> bool: ... + + def normalize(self, row: SpanRow, attributes: Mapping[str, str]) -> None: ... + + +def to_int(value: str | None) -> int: + try: + return int(value) if value else 0 + except ValueError: + return 0 diff --git a/litellm/tracing/normalizers/genai.py b/litellm/tracing/normalizers/genai.py new file mode 100644 index 00000000000..16986607396 --- /dev/null +++ b/litellm/tracing/normalizers/genai.py @@ -0,0 +1,31 @@ +from collections.abc import Mapping +from dataclasses import dataclass +from typing import Final + +from litellm.tracing.types import SpanRow + +_LLM_OPERATIONS: Final = frozenset({"chat", "text_completion", "generate_content"}) + + +@dataclass(frozen=True, slots=True) +class GenAISemconvNormalizer: + """OTEL `gen_ai.*` semantic conventions. Matches every span, so it belongs last as the fallback.""" + + name: str = "genai" + + def matches(self, scope_name: str, attributes: Mapping[str, str]) -> bool: + return True + + def normalize(self, row: SpanRow, attributes: Mapping[str, str]) -> None: + operation: Final = attributes.get("gen_ai.operation.name", "") + if operation == "invoke_agent" or not row["ParentSpanId"]: + row["ObservationType"] = "agent" + elif operation in _LLM_OPERATIONS: + row["ObservationType"] = "llm" + elif operation == "execute_tool": + row["ObservationType"] = "tool" + row["AgentName"] = attributes.get("gen_ai.agent.name", "") + row["Model"] = attributes.get("gen_ai.request.model") or attributes.get("gen_ai.response.model", "") + row["LiteLLMRequestId"] = attributes.get("gen_ai.response.id", "") + row["Input"] = attributes.get("gen_ai.input.messages") or attributes.get("gen_ai.tool.call.arguments", "") + row["Output"] = attributes.get("gen_ai.output.messages") or attributes.get("gen_ai.tool.call.result", "") diff --git a/litellm/tracing/normalizers/langsmith.py b/litellm/tracing/normalizers/langsmith.py new file mode 100644 index 00000000000..daca932d57d --- /dev/null +++ b/litellm/tracing/normalizers/langsmith.py @@ -0,0 +1,115 @@ +"""LangSmith OTEL mode, which LangChain, LangGraph and Deep Agents export through.""" + +import json +from collections.abc import Mapping +from dataclasses import dataclass +from types import MappingProxyType +from typing import Final + +from litellm.tracing.normalizers.messages import lc_message +from litellm.tracing.types import SpanRow, SpanType + +# LangChain / Deep Agents middleware wrappers: real spans, but noise in the UI +_FRAMEWORK_SUFFIXES: Final = ( + ".wrap_model_call", + ".wrap_tool_call", + ".before_agent", + ".after_agent", + ".before_model", + ".after_model", +) + + +def _loads(value: str) -> object: + try: + return json.loads(value) + except (ValueError, TypeError): + return None + + +def _span_type(row: SpanRow, attributes: Mapping[str, str]) -> SpanType: + kind: Final = attributes.get("langsmith.span.kind", "chain") + name: Final = row["SpanName"] + if not row["ParentSpanId"] or name == attributes.get("langsmith.metadata.lc_agent_name"): + return "agent" + if kind in ("llm", "tool"): + return kind + if name.endswith(_FRAMEWORK_SUFFIXES): + return "framework" + return "chain" + + +def _tool_output(completion: object) -> object: + raw: Final = completion.get("output", completion) if isinstance(completion, dict) else completion + update: Final = raw.get("update") if isinstance(raw, dict) else None + update_messages: Final = update.get("messages") or () if isinstance(update, dict) else () + is_command: Final = isinstance(raw, dict) and "update" in raw + # LangGraph Command (e.g. the Deep Agents `task` tool): the result is the last update message + output: Final = update_messages[-1] if is_command and update_messages else raw + return output.get("content", output) if isinstance(output, dict) else output + + +def _set_agent_io(row: SpanRow, attributes: Mapping[str, str], prompt: object, completion: object) -> None: + input_messages: Final = prompt.get("messages") if isinstance(prompt, dict) else None + output_messages: Final = completion.get("messages") if isinstance(completion, dict) else None + # agents built with @traceable take arbitrary args, not a message list: keep the raw payload then + row["Input"] = ( + json.dumps(tuple(lc_message(m) for m in input_messages if isinstance(m, dict))) + if input_messages + else attributes.get("gen_ai.prompt", "") + ) + row["Output"] = ( + json.dumps(lc_message(output_messages[-1])) + if output_messages and isinstance(output_messages[-1], dict) + else attributes.get("gen_ai.completion", "") + ) + + +def _set_io(row: SpanRow, attributes: Mapping[str, str]) -> None: + prompt: Final = _loads(attributes.get("gen_ai.prompt", "")) + completion: Final = _loads(attributes.get("gen_ai.completion", "")) + if row["ObservationType"] == "llm" and isinstance(completion, dict): + prompt_payload: Final = prompt if isinstance(prompt, dict) else MappingProxyType({}) + messages: Final = prompt_payload.get("messages") or ((),) + batch: Final = messages[0] if messages and isinstance(messages[0], list) else messages + row["Input"] = ( + json.dumps(tuple(lc_message(m) for m in batch if isinstance(m, dict))) + if isinstance(batch, (list, tuple)) + else "" + ) + generations: Final = completion.get("generations") + first: Final = generations[0] if isinstance(generations, list) and generations else None + item: Final = first[0] if isinstance(first, list) and first else None + message: Final = item.get("message") if isinstance(item, dict) else None + generation: Final = message.get("kwargs") if isinstance(message, dict) else None + if isinstance(generation, dict): + row["Output"] = json.dumps(lc_message(generation)) + metadata: Final = generation.get("response_metadata") + row["LiteLLMRequestId"] = metadata.get("id", "") if isinstance(metadata, dict) else "" + return + row["Output"] = attributes.get("gen_ai.completion", "") + return + if row["ObservationType"] == "tool": + output: Final = _tool_output(completion) + row["Input"] = attributes.get("gen_ai.prompt", "") + row["Output"] = output if isinstance(output, str) else json.dumps(output) + return + if row["ObservationType"] == "agent": + _set_agent_io(row, attributes, prompt, completion) + return + row["Input"] = attributes.get("gen_ai.prompt", "") + row["Output"] = attributes.get("gen_ai.completion", "") + + +@dataclass(frozen=True, slots=True) +class LangSmithNormalizer: + name: str = "langsmith" + + def matches(self, scope_name: str, attributes: Mapping[str, str]) -> bool: + return scope_name == "langsmith" or "langsmith.span.kind" in attributes + + def normalize(self, row: SpanRow, attributes: Mapping[str, str]) -> None: + row["ObservationType"] = _span_type(row, attributes) + row["AgentName"] = attributes.get("langsmith.metadata.lc_agent_name", "") + row["Model"] = attributes.get("gen_ai.request.model", "") + _set_io(row, attributes) diff --git a/litellm/tracing/normalizers/messages.py b/litellm/tracing/normalizers/messages.py new file mode 100644 index 00000000000..0d552d05b82 --- /dev/null +++ b/litellm/tracing/normalizers/messages.py @@ -0,0 +1,59 @@ +import json +from collections.abc import Mapping +from types import MappingProxyType +from typing import Any, Final, Literal, TypeAlias + +from pydantic import BaseModel, ConfigDict, TypeAdapter, ValidationError + +ChatRole: TypeAlias = Literal["system", "user", "assistant", "tool"] + +MESSAGE_ROLES: Final[Mapping[str, ChatRole]] = MappingProxyType( + {"human": "user", "user": "user", "ai": "assistant", "assistant": "assistant", "system": "system", "tool": "tool"} +) + + +class _ContentBlock(BaseModel): + model_config = ConfigDict(frozen=True, extra="ignore") + type: str = "" + text: str | None = None + + +_CONTENT_BLOCKS: Final = TypeAdapter(tuple[_ContentBlock, ...]) +_NON_TEXT_BLOCKS: Final = frozenset( + {"reasoning", "thinking", "redacted_thinking", "function_call", "tool_use", "tool_call"} +) + + +def content_text(content: object) -> str: + """Message content as display text: Responses-style block lists keep only their text blocks.""" + if content is None: + return "" + if isinstance(content, str): + return content + try: + blocks: Final = _CONTENT_BLOCKS.validate_python(content) + except ValidationError: + return json.dumps(content) + if not all(block.text is not None or block.type in _NON_TEXT_BLOCKS for block in blocks): + return json.dumps(content) + return "\n\n".join(block.text for block in blocks if block.text is not None) + + +def lc_message(message: Mapping[str, Any]) -> dict[str, Any]: + """LangChain serialized message (or plain {role, content}) -> {role, content, tool_calls?}.""" + kwargs: Final = message.get("kwargs", message) + role: Final = MESSAGE_ROLES.get( + kwargs.get("type") or kwargs.get("role"), kwargs.get("role") or kwargs.get("type") or "" + ) + out: Final[dict[str, Any]] = { # mutable-ok: the framework message is built for JSON serialization + "role": role, + "content": content_text(kwargs.get("content", "")), + } + if kwargs.get("tool_calls"): + out["tool_calls"] = tuple( + {"name": t.get("name"), "args": t.get("args")} # mutable-ok: JSON tool calls need object payloads + for t in kwargs["tool_calls"] + ) + if role == "tool" and kwargs.get("name"): + out["name"] = kwargs["name"] + return out diff --git a/litellm/tracing/normalizers/openinference.py b/litellm/tracing/normalizers/openinference.py new file mode 100644 index 00000000000..f9e1295148c --- /dev/null +++ b/litellm/tracing/normalizers/openinference.py @@ -0,0 +1,27 @@ +from collections.abc import Mapping +from dataclasses import dataclass +from types import MappingProxyType +from typing import Final + +from litellm.tracing.normalizers.base import to_int +from litellm.tracing.types import SpanRow, SpanType + +_OPENINFERENCE_TYPES: Final[Mapping[str, SpanType]] = MappingProxyType({"AGENT": "agent", "LLM": "llm", "TOOL": "tool"}) + + +@dataclass(frozen=True, slots=True) +class OpenInferenceNormalizer: + name: str = "openinference" + + def matches(self, scope_name: str, attributes: Mapping[str, str]) -> bool: + return "openinference.span.kind" in attributes + + def normalize(self, row: SpanRow, attributes: Mapping[str, str]) -> None: + kind: Final = attributes.get("openinference.span.kind", "").upper() + row["ObservationType"] = _OPENINFERENCE_TYPES.get(kind, "agent" if not row["ParentSpanId"] else "chain") + row["AgentName"] = attributes.get("agent.name", "") + row["Model"] = attributes.get("llm.model_name", "") + row["Input"] = attributes.get("input.value", "") + row["Output"] = attributes.get("output.value", "") + row["InputTokens"] = to_int(attributes.get("llm.token_count.prompt")) + row["OutputTokens"] = to_int(attributes.get("llm.token_count.completion")) diff --git a/litellm/tracing/store.py b/litellm/tracing/store.py index 806757306c0..fdb1c7820f2 100644 --- a/litellm/tracing/store.py +++ b/litellm/tracing/store.py @@ -28,6 +28,7 @@ from litellm.tracing.types import ( TraceScope, TraceSummary, ) +from litellm.tracing.ui_format import to_ui_content NANOS_PER_MS: Final = 1_000_000 SPEND_WINDOW_MS: Final = 30 * 60 * 1000 @@ -344,5 +345,7 @@ class ClickHouseTraceStore: span_id=rows[0]["span_id"], input=rows[0]["input"], output=rows[0]["output"], + input_ui=to_ui_content(rows[0]["input"]), + output_ui=to_ui_content(rows[0]["output"]), attributes=rows[0]["attributes"], ) diff --git a/litellm/tracing/types.py b/litellm/tracing/types.py index 6cdfcd84da7..fcf0d83fb8f 100644 --- a/litellm/tracing/types.py +++ b/litellm/tracing/types.py @@ -14,6 +14,8 @@ from typing import Literal from typing_extensions import NotRequired, ReadOnly, TypedDict +from litellm.tracing.ui_format import UIContent + SpanType = Literal["agent", "llm", "tool", "chain", "framework"] SpanStatus = Literal["ok", "error", "unset"] @@ -84,6 +86,8 @@ class SpanDetail(TypedDict): span_id: ReadOnly[str] input: ReadOnly[str] output: ReadOnly[str] + input_ui: ReadOnly[UIContent] + output_ui: ReadOnly[UIContent] attributes: ReadOnly[dict[str, str]] diff --git a/litellm/tracing/ui_format.py b/litellm/tracing/ui_format.py new file mode 100644 index 00000000000..d7ecf48078f --- /dev/null +++ b/litellm/tracing/ui_format.py @@ -0,0 +1,158 @@ +"""The LiteLLM UI content format: span input / output reduced to messages, key/value fields or plain text.""" + +import json +from collections.abc import Mapping, Sequence +from typing import Final, Literal, TypeAlias + +from pydantic import BaseModel, ConfigDict, JsonValue, TypeAdapter, ValidationError +from typing_extensions import NotRequired, ReadOnly, TypedDict + +from litellm.tracing.normalizers.messages import MESSAGE_ROLES, ChatRole, content_text + + +class UIToolCall(TypedDict): + name: ReadOnly[str] + arguments: ReadOnly[str] + + +class UIMessage(TypedDict): + role: ReadOnly[ChatRole] + content: ReadOnly[str] + name: ReadOnly[NotRequired[str]] + tool_calls: ReadOnly[NotRequired[tuple[UIToolCall, ...]]] + + +class UIField(TypedDict): + key: ReadOnly[str] + value: ReadOnly[str] + + +class UIMessages(TypedDict): + kind: ReadOnly[Literal["messages"]] + messages: ReadOnly[tuple[UIMessage, ...]] + + +class UIFields(TypedDict): + kind: ReadOnly[Literal["fields"]] + fields: ReadOnly[tuple[UIField, ...]] + + +class UIText(TypedDict): + kind: ReadOnly[Literal["text"]] + text: ReadOnly[str] + + +UIContent: TypeAlias = UIMessages | UIFields | UIText + + +class _ToolFunction(BaseModel): + model_config = ConfigDict(frozen=True, extra="ignore") + name: str = "" + arguments: JsonValue = None + + +class _RawToolCall(BaseModel): + model_config = ConfigDict(frozen=True, extra="ignore") + name: str = "" + args: JsonValue = None + arguments: JsonValue = None + function: _ToolFunction | None = None + + +class _RawMessage(BaseModel): + model_config = ConfigDict(frozen=True, extra="ignore") + role: str | None = None + type: str | None = None + content: JsonValue = None + name: str | None = None + tool_calls: tuple[_RawToolCall, ...] | None = None + kwargs: "_RawMessage | None" = None + + +_JSON: Final[TypeAdapter[JsonValue]] = TypeAdapter(JsonValue) +_MESSAGE: Final = TypeAdapter(_RawMessage) +_MESSAGES: Final = TypeAdapter(tuple[_RawMessage, ...]) + + +def _unwrapped(message: _RawMessage) -> _RawMessage: + return message.kwargs if message.kwargs is not None else message + + +def _is_message(message: _RawMessage) -> bool: + has_role: Final = message.role is not None or message.type in MESSAGE_ROLES + return has_role and ("content" in message.model_fields_set or bool(message.tool_calls)) + + +def _arguments_text(arguments: JsonValue) -> str: + match arguments: + case str(): + return arguments + case None: + return "{}" + case _: + return json.dumps(arguments) + + +def _tool_call(call: _RawToolCall) -> UIToolCall: + if call.function is not None: + return UIToolCall(name=call.function.name or call.name, arguments=_arguments_text(call.function.arguments)) + return UIToolCall(name=call.name, arguments=_arguments_text(call.arguments if call.args is None else call.args)) + + +def _role(message: _RawMessage, has_tool_calls: bool) -> ChatRole: + """Known roles and LangChain types map directly; any other role is the assistant when it calls tools, else the user.""" + known: Final = MESSAGE_ROLES.get(message.role or message.type or "") + if known is not None: + return known + return "assistant" if has_tool_calls else "user" + + +def _ui_message(message: _RawMessage) -> UIMessage: + calls: Final = tuple(_tool_call(call) for call in message.tool_calls or ()) + role: Final = _role(message, bool(calls)) + content: Final = content_text(message.content) + match (message.name or None, calls): + case (None, ()): + return UIMessage(role=role, content=content) + case (None, _): + return UIMessage(role=role, content=content, tool_calls=calls) + case (str() as name, ()): + return UIMessage(role=role, content=content, name=name) + case (str() as name, _): + return UIMessage(role=role, content=content, name=name, tool_calls=calls) + + +def _messages(parsed: Sequence[JsonValue] | Mapping[str, JsonValue]) -> tuple[_RawMessage, ...] | None: + try: + raw: Final = ( + (_MESSAGE.validate_python(parsed),) if isinstance(parsed, Mapping) else _MESSAGES.validate_python(parsed) + ) + except ValidationError: + return None + unwrapped: Final = tuple(_unwrapped(message) for message in raw) + return unwrapped if unwrapped and all(_is_message(message) for message in unwrapped) else None + + +def _field_value(value: JsonValue) -> str: + return value if isinstance(value, str) else json.dumps(value) + + +def _parsed(raw: str) -> JsonValue: + try: + return _JSON.validate_json(raw) + except ValidationError: + return raw + + +def to_ui_content(raw: str) -> UIContent: + if not raw: + return UIText(kind="text", text="") + parsed: Final = _parsed(raw) + if not isinstance(parsed, list | dict): + return UIText(kind="text", text=parsed if isinstance(parsed, str) else raw) + messages: Final = _messages(parsed) + if messages is not None: + return UIMessages(kind="messages", messages=tuple(_ui_message(message) for message in messages)) + if isinstance(parsed, dict): + return UIFields(kind="fields", fields=tuple(UIField(key=k, value=_field_value(v)) for k, v in parsed.items())) + return UIText(kind="text", text=raw) diff --git a/tests/test_litellm/proxy/test_tracing_endpoints.py b/tests/test_litellm/proxy/test_tracing_endpoints.py index 4c7c70a39f3..6391c1577f3 100644 --- a/tests/test_litellm/proxy/test_tracing_endpoints.py +++ b/tests/test_litellm/proxy/test_tracing_endpoints.py @@ -11,7 +11,8 @@ from fastapi.testclient import TestClient from litellm.proxy import tracing_endpoints from litellm.proxy._types import LitellmUserRoles, UserAPIKeyAuth from litellm.proxy.auth.user_api_key_auth import user_api_key_auth -from litellm.tracing import TracingPayloadTooLargeError +from litellm.tracing import TraceReceiver, TracingPayloadTooLargeError +from litellm.tracing.store import ClickHouseTraceStore TEAM_KEY = UserAPIKeyAuth( token="hashed-key", team_id="team-research", org_id="org-1", user_role=LitellmUserRoles.INTERNAL_USER @@ -148,6 +149,24 @@ def test_get_span_404_and_200(client, receiver): receiver.get_span.assert_awaited_with("t1", "s1", {"team_ids": ("team-research",), "api_key_hash": ""}, "") +def test_get_span_serves_ui_content_from_stored_payloads(client, monkeypatch): + storage = MagicMock() + stored_output = '{"role": "ai", "content": "", "tool_calls": [{"name": "lookup", "args": {"id": 7}}]}' + storage.query = AsyncMock( + return_value=[{"span_id": "s1", "input": '{"city": "Paris"}', "output": stored_output, "attributes": {}}] + ) + monkeypatch.setattr(tracing_endpoints, "receiver", TraceReceiver(ClickHouseTraceStore(storage))) + body = client.get("/v1/traces/t1/spans/s1").json() + assert body["output"] == stored_output + assert body["input_ui"] == {"kind": "fields", "fields": [{"key": "city", "value": "Paris"}]} + assert body["output_ui"] == { + "kind": "messages", + "messages": [ + {"role": "assistant", "content": "", "tool_calls": [{"name": "lookup", "arguments": '{"id": 7}'}]} + ], + } + + def test_trace_detail_passes_scoped_reference(client, receiver): receiver.get_trace.return_value = {"summary": {"trace_id": "t1"}, "agents": [], "spans": []} assert client.get("/v1/traces/t1?trace_ref=run-one").status_code == 200 diff --git a/tests/test_litellm/tracing/normalizers/test_registry.py b/tests/test_litellm/tracing/normalizers/test_registry.py new file mode 100644 index 00000000000..4c2fd051d6c --- /dev/null +++ b/tests/test_litellm/tracing/normalizers/test_registry.py @@ -0,0 +1,68 @@ +from collections.abc import Mapping +from dataclasses import dataclass +from types import MappingProxyType +from typing import Final + +from litellm.tracing.normalizers import ( + NORMALIZERS, + GenAISemconvNormalizer, + LangSmithNormalizer, + OpenInferenceNormalizer, + select_normalizer, +) +from litellm.tracing.types import SpanRow + +_NO_ATTRIBUTES: Final[Mapping[str, str]] = MappingProxyType({}) + + +def test_langsmith_scope_selects_langsmith_without_any_attributes(): + assert isinstance(select_normalizer("langsmith", _NO_ATTRIBUTES), LangSmithNormalizer) + + +def test_langsmith_kind_attribute_selects_langsmith_under_any_scope(): + assert isinstance(select_normalizer("other", MappingProxyType({"langsmith.span.kind": "llm"})), LangSmithNormalizer) + + +def test_langsmith_wins_over_openinference_when_both_markers_present(): + attributes: Final = MappingProxyType({"langsmith.span.kind": "llm", "openinference.span.kind": "LLM"}) + assert isinstance(select_normalizer("other", attributes), LangSmithNormalizer) + + +def test_openinference_kind_attribute_selects_openinference(): + assert isinstance( + select_normalizer("other", MappingProxyType({"openinference.span.kind": "LLM"})), OpenInferenceNormalizer + ) + + +def test_unmarked_span_falls_back_to_genai(): + assert isinstance( + select_normalizer("other", MappingProxyType({"gen_ai.operation.name": "chat"})), GenAISemconvNormalizer + ) + + +def test_empty_registry_falls_back_to_genai(): + assert isinstance(select_normalizer("langsmith", _NO_ATTRIBUTES, registry=()), GenAISemconvNormalizer) + + +def test_registry_names_are_unique(): + names: Final = tuple(n.name for n in NORMALIZERS) + assert len(names) == len(frozenset(names)) + + +@dataclass(frozen=True, slots=True) +class _CustomNormalizer: + name: str = "custom" + + def matches(self, scope_name: str, attributes: Mapping[str, str]) -> bool: + return scope_name == "custom-sdk" + + def normalize(self, row: SpanRow, attributes: Mapping[str, str]) -> None: + return None + + +def test_normalizer_inserted_ahead_in_custom_registry_wins_only_where_it_matches(): + registry: Final = (_CustomNormalizer(), *NORMALIZERS) + assert isinstance( + select_normalizer("custom-sdk", MappingProxyType({"langsmith.span.kind": "llm"}), registry), _CustomNormalizer + ) + assert isinstance(select_normalizer("langsmith", _NO_ATTRIBUTES, registry), LangSmithNormalizer) diff --git a/tests/test_litellm/tracing/test_decode.py b/tests/test_litellm/tracing/test_decode.py index 168ff2bf7fb..7185341d038 100644 --- a/tests/test_litellm/tracing/test_decode.py +++ b/tests/test_litellm/tracing/test_decode.py @@ -125,6 +125,49 @@ def test_incomplete_langsmith_completion_preserves_the_export(completion): assert rows[0]["Output"] == completion +def test_llm_block_list_content_keeps_only_text(): + reasoning = {"type": "reasoning", "summary": [], "encrypted_content": "gAAAAB-opaque"} + history = [reasoning, {"type": "text", "text": "Earlier answer", "annotations": []}] + answer = [reasoning, {"type": "text", "text": "Part one"}, {"type": "text", "text": "Part two"}] + prompt = { + "messages": [ + [ + {"kwargs": {"type": "human", "content": "refund please"}}, + {"kwargs": {"type": "ai", "content": history}}, + {"kwargs": {"type": "ai", "content": [reasoning]}}, + ] + ] + } + completion = {"generations": [[{"message": {"kwargs": {"type": "ai", "content": answer}}}]]} + span = _span( + "ChatOpenAI", + b"\x03" * 8, + b"\x02" * 8, + langsmith__span__kind="llm", + gen_ai__prompt=json.dumps(prompt), + gen_ai__completion=json.dumps(completion), + ) + rows = decode_otlp(_export(span, scope="langsmith"), "application/x-protobuf") + assert [m["content"] for m in json.loads(rows[0]["Input"])] == ["refund please", "Earlier answer", ""] + assert json.loads(rows[0]["Output"])["content"] == "Part one\n\nPart two" + assert "encrypted_content" not in rows[0]["Input"] + rows[0]["Output"] + + +def test_llm_unrecognized_list_content_is_kept_as_json(): + content = [{"type": "image_url", "image_url": {"url": "https://x.test/a.png"}}] + completion = {"generations": [[{"message": {"kwargs": {"type": "ai", "content": content}}}]]} + span = _span( + "ChatOpenAI", + b"\x03" * 8, + b"\x02" * 8, + langsmith__span__kind="llm", + gen_ai__prompt='{"messages": [[{"kwargs": {"type": "human", "content": "hi"}}]]}', + gen_ai__completion=json.dumps(completion), + ) + rows = decode_otlp(_export(span, scope="langsmith"), "application/x-protobuf") + assert json.loads(json.loads(rows[0]["Output"])["content"]) == content + + def test_task_tool_output_is_subagent_final_message_text(rows_by_name): task = rows_by_name["task"] assert json.loads(task["Input"])["subagent_type"] == "researcher" @@ -189,6 +232,64 @@ def test_long_values_are_truncated_with_marker(): assert len(task["Input"].split("…")[0].encode()) <= 100 +def test_long_message_history_drops_middle_messages_and_stays_valid_json(): + history = [{"kwargs": {"type": "human", "content": f"turn {i} " + "x" * 60}} for i in range(12)] + prompt = json.dumps({"messages": [[{"kwargs": {"type": "system", "content": "be brief"}}, *history]]}) + completion = json.dumps({"generations": [[{"message": {"kwargs": {"type": "ai", "content": "ok"}}}]]}) + span = _span( + "ChatOpenAI", + b"\x03" * 8, + b"\x02" * 8, + langsmith__span__kind="llm", + gen_ai__prompt=prompt, + gen_ai__completion=completion, + ) + with patch.object(decode, "OTLP_MAX_ATTRIBUTE_VALUE_BYTES", 400): + rows = decode_otlp(_export(span, scope="langsmith"), "application/x-protobuf") + messages = json.loads(rows[0]["Input"]) + assert len(rows[0]["Input"].encode()) <= 400 + assert messages[0]["content"] == "be brief" + assert "earlier messages truncated" in messages[1]["content"] + assert messages[-1]["content"].startswith("turn 11 ") + kept = int(messages[1]["content"].split("[")[1].split()[0]) + assert kept + len(messages) - 2 == 12 + + +@pytest.mark.parametrize( + "messages", + [ + [{"role": "system", "content": "s" * 2000}, {"role": "user", "content": "short question"}], + [{"role": "user", "content": "a" * 900}, {"role": "assistant", "content": "b" * 900}], + [ + {"role": "system", "content": "s" * 900}, + {"role": "user", "content": "middle"}, + {"role": "user", "content": "q" * 900}, + ], + ], + ids=["huge-first-message", "two-messages", "huge-first-and-last"], +) +def test_oversized_message_arrays_are_shortened_not_cut(messages): + with patch.object(decode, "OTLP_MAX_ATTRIBUTE_VALUE_BYTES", 400): + out = decode._truncate_payload(json.dumps(messages)) + assert len(out.encode()) <= 400 + kept = json.loads(out) + assert kept[0]["role"] == messages[0]["role"] + assert kept[-1]["role"] == messages[-1]["role"] + assert all(isinstance(m["content"], str) for m in kept) + + +def test_oversized_non_content_fields_still_fit_the_limit(): + heavy = {"role": "assistant", "content": "x", "tool_calls": [{"name": "t", "args": {"blob": "z" * 3000}}]} + messages = [heavy, {"role": "user", "content": "—" * 900}] + with patch.object(decode, "OTLP_MAX_ATTRIBUTE_VALUE_BYTES", 400): + out = decode._truncate_payload(json.dumps(messages)) + kept = json.loads(out) + assert len(out.encode()) <= 400 + assert [m["role"] for m in kept] == ["assistant", "user"] + assert kept[0]["content"].startswith("x") + assert kept[1]["content"].startswith("\u2014") + + # ---------------------------------------------------------------- status / exceptions diff --git a/tests/test_litellm/tracing/test_store.py b/tests/test_litellm/tracing/test_store.py index 7ee772e078c..2e80594ff9b 100644 --- a/tests/test_litellm/tracing/test_store.py +++ b/tests/test_litellm/tracing/test_store.py @@ -306,11 +306,16 @@ async def test_get_span_not_found_and_found(): store = ClickHouseTraceStore(client) scope: TraceScope = {"team_ids": (), "api_key_hash": ""} assert await store.get_span("t", "s", scope) is None - client.query = AsyncMock(return_value=[{"span_id": "s", "input": "i", "output": "o", "attributes": {"k": "v"}}]) + stored_input = '[{"role": "user", "content": "hi"}]' + client.query = AsyncMock( + return_value=[{"span_id": "s", "input": stored_input, "output": '{"ok": true}', "attributes": {"k": "v"}}] + ) assert await store.get_span("t", "s", scope) == { "span_id": "s", - "input": "i", - "output": "o", + "input": stored_input, + "output": '{"ok": true}', + "input_ui": {"kind": "messages", "messages": ({"role": "user", "content": "hi"},)}, + "output_ui": {"kind": "fields", "fields": ({"key": "ok", "value": "true"},)}, "attributes": {"k": "v"}, } diff --git a/tests/test_litellm/tracing/test_ui_format.py b/tests/test_litellm/tracing/test_ui_format.py new file mode 100644 index 00000000000..27c554041ef --- /dev/null +++ b/tests/test_litellm/tracing/test_ui_format.py @@ -0,0 +1,119 @@ +import json + +import pytest + +from litellm.tracing.ui_format import to_ui_content + + +def test_message_array_maps_roles_and_keeps_order(): + raw = json.dumps( + [ + {"role": "system", "content": "be brief"}, + {"role": "human", "content": "hi"}, + {"role": "tool", "name": "lookup", "content": "42"}, + {"role": "narrator", "content": "aside"}, + ] + ) + assert to_ui_content(raw) == { + "kind": "messages", + "messages": ( + {"role": "system", "content": "be brief"}, + {"role": "user", "content": "hi"}, + {"role": "tool", "content": "42", "name": "lookup"}, + {"role": "user", "content": "aside"}, + ), + } + + +@pytest.mark.parametrize( + "call", + [ + {"name": "get_plan", "args": {"customer_id": "c-1"}}, + {"name": "get_plan", "arguments": '{"customer_id": "c-1"}'}, + {"id": "call_1", "type": "function", "function": {"name": "get_plan", "arguments": '{"customer_id": "c-1"}'}}, + ], +) +def test_single_assistant_message_with_tool_call(call: dict[str, object]): + content = to_ui_content(json.dumps({"role": "assistant", "content": None, "tool_calls": [call]})) + assert content["kind"] == "messages" + (message,) = content["messages"] + assert message["role"] == "assistant" + assert message["content"] == "" + calls = message.get("tool_calls") + assert calls is not None and len(calls) == 1 + assert calls[0]["name"] == "get_plan" + assert json.loads(calls[0]["arguments"]) == {"customer_id": "c-1"} + + +def test_unknown_role_with_tool_calls_is_assistant(): + content = to_ui_content(json.dumps({"role": "model", "content": "", "tool_calls": [{"name": "f", "args": None}]})) + assert content == { + "kind": "messages", + "messages": ({"role": "assistant", "content": "", "tool_calls": ({"name": "f", "arguments": "{}"},)},), + } + + +def test_block_list_content_keeps_text_and_drops_reasoning(): + raw = json.dumps( + { + "role": "assistant", + "content": [ + {"type": "reasoning", "encrypted_content": "opaque"}, + {"type": "thinking", "thinking": "hidden chain"}, + {"type": "text", "text": "first"}, + {"type": "text", "text": "second"}, + ], + } + ) + assert to_ui_content(raw) == { + "kind": "messages", + "messages": ({"role": "assistant", "content": "first\n\nsecond"},), + } + + +def test_langchain_kwargs_shape(): + raw = json.dumps( + [ + {"lc": 1, "type": "constructor", "kwargs": {"type": "human", "content": "question"}}, + {"kwargs": {"type": "ai", "content": "", "tool_calls": [{"name": "search", "args": {"q": "x"}}]}}, + ] + ) + content = to_ui_content(raw) + assert content["kind"] == "messages" + human, ai = content["messages"] + assert human == {"role": "user", "content": "question"} + assert ai["role"] == "assistant" + assert ai.get("tool_calls") == ({"name": "search", "arguments": '{"q": "x"}'},) + + +def test_plain_object_becomes_fields_in_key_order(): + raw = json.dumps({"zeta": "plain", "alpha": {"nested": [1, 2]}, "count": 3, "missing": None}) + assert to_ui_content(raw) == { + "kind": "fields", + "fields": ( + {"key": "zeta", "value": "plain"}, + {"key": "alpha", "value": '{"nested": [1, 2]}'}, + {"key": "count", "value": "3"}, + {"key": "missing", "value": "null"}, + ), + } + + +def test_object_with_role_but_no_content_is_fields(): + assert to_ui_content('{"role": "admin", "user_id": "u1"}')["kind"] == "fields" + + +def test_json_string_becomes_its_text(): + assert to_ui_content(json.dumps('line one\n"quoted"')) == {"kind": "text", "text": 'line one\n"quoted"'} + + +@pytest.mark.parametrize( + "raw", + ['[{"role": "user", "content": "cut of', "plain words", "42", "[1, 2]", "[]"], +) +def test_non_message_non_object_payloads_keep_the_raw_string(raw: str): + assert to_ui_content(raw) == {"kind": "text", "text": raw} + + +def test_empty_is_empty_text(): + assert to_ui_content("") == {"kind": "text", "text": ""} diff --git a/ui/litellm-dashboard/src/app/globals.css b/ui/litellm-dashboard/src/app/globals.css index 87959ca0139..e1ff921648f 100644 --- a/ui/litellm-dashboard/src/app/globals.css +++ b/ui/litellm-dashboard/src/app/globals.css @@ -64,6 +64,61 @@ } @theme inline { + --animate-slide-left: slide-left 200ms cubic-bezier(0, 0, 0.2, 1) both; + --animate-view-fade-in: view-fade-in 100ms cubic-bezier(0.4, 0, 0.2, 1) both; + --animate-slot-slide-in: slot-slide-in 150ms cubic-bezier(0, 0, 0.2, 1) both; + --animate-trace-drawer-in: trace-drawer-in 200ms cubic-bezier(0.25, 1, 0.5, 1) both; + --animate-trace-drawer-out: trace-drawer-out 200ms cubic-bezier(0.4, 0, 1, 1) both; + + @keyframes trace-drawer-in { + from { + opacity: 0; + transform: translateX(2rem) scaleX(0.98); + } + to { + opacity: 1; + transform: none; + } + } + @keyframes trace-drawer-out { + from { + opacity: 1; + transform: none; + } + to { + opacity: 0; + transform: translateX(2rem) scaleX(0.98); + } + } + + @keyframes slide-left { + from { + opacity: 0; + transform: translateX(2rem) scaleX(0.98); + } + to { + opacity: 1; + transform: none; + } + } + @keyframes view-fade-in { + from { + opacity: 0; + } + to { + opacity: 1; + } + } + @keyframes slot-slide-in { + from { + opacity: 0; + transform: translateY(4px); + } + to { + opacity: 1; + transform: none; + } + } @keyframes scroll-fade-reveal-e { from { --scroll-fade-e: var(--_scroll-fade-size-e, var(--scroll-fade-size, min(12%, calc(var(--spacing) * 10)))); @@ -74,6 +129,22 @@ } } +@layer utilities { + .animate-trace-drawer-in, + .animate-trace-drawer-out { + transform-origin: right center; + } + @media (prefers-reduced-motion: reduce) { + .animate-trace-drawer-in, + .animate-trace-drawer-out, + .animate-slide-left, + .animate-view-fade-in, + .animate-slot-slide-in { + animation: none !important; + } + } +} + @utility scroll-fade-e { --_scroll-fade-size-e: var(--scroll-fade-e-size, var(--scroll-fade-size, min(12%, calc(var(--spacing) * 10)))); --scroll-fade-mask: linear-gradient(to right, #000 0, #000 calc(100% - var(--scroll-fade-e, 0px)), transparent 100%); @@ -141,6 +212,38 @@ --sidebar-ring: oklch(0.707 0.022 261.325); --neutral-border: #dcddeb; --logo-surface: oklch(1 0 0); + --trace-text: oklch(0.21 0.03 256); + --trace-text-2: oklch(0.35 0.03 256); + --trace-text-secondary: oklch(0.35 0.03 256); + --trace-duration: oklch(0.48 0.03 230); + --trace-key: oklch(0.55 0.03 240); + --trace-placeholder: oklch(0.7 0.02 240); + --trace-surface: oklch(1 0 0); + --trace-chip: oklch(0.975 0.006 220); + --trace-row-hover: oklch(0.975 0.008 215); + --trace-row-selected: oklch(0.95 0.035 200); + --trace-brand: oklch(0.6 0.13 195); + --trace-border: oklch(0.92 0.01 230); + --trace-line: oklch(0.88 0.03 205); + --trace-card-border: oklch(0.93 0.01 230); + --trace-dot: oklch(0.86 0.05 190); + --trace-tab-active: oklch(0.95 0.025 205); + --trace-tab-hover: oklch(0.93 0.02 215); + --trace-tag: oklch(0.95 0.02 205); + --trace-chain: oklch(0.56 0.17 255); + --trace-llm: oklch(0.6 0.13 215); + --trace-tool: oklch(0.64 0.14 165); + --trace-glyph: oklch(0.99 0 0); + --trace-human: oklch(0.5 0.15 260); + --trace-human-glyph: oklch(0.95 0.04 210); + --trace-turn: oklch(0.96 0.03 200); + --trace-turn-border: oklch(0.75 0.1 200); + --trace-ok: oklch(0.92 0.08 160); + --trace-ok-glyph: oklch(0.55 0.15 155); + --trace-called: oklch(0.93 0.06 185); + --trace-called-text: oklch(0.38 0.08 195); + --trace-shadow-md: 0 4px 6px -1px #0b1b2e1a, 0 2px 4px -1px #0b1b2e0f; + --trace-shadow-xs: 0 1px 2px 0 #0b1b2e0d; } .dark { @@ -183,9 +286,73 @@ --sidebar-border: oklch(0.187 0 0); --sidebar-ring: oklch(0.569 0 0); --neutral-border: var(--border); + --trace-text: oklch(0.96 0.005 220); + --trace-text-2: oklch(0.88 0.01 220); + --trace-text-secondary: oklch(0.88 0.01 220); + --trace-duration: oklch(0.78 0.03 200); + --trace-key: oklch(0.68 0.03 220); + --trace-placeholder: oklch(0.5 0.02 230); + --trace-surface: oklch(0.19 0.012 240); + --trace-chip: oklch(0.23 0.015 235); + --trace-row-hover: oklch(0.23 0.018 230); + --trace-row-selected: oklch(0.29 0.05 210); + --trace-brand: oklch(0.78 0.13 190); + --trace-border: oklch(0.3 0.02 235); + --trace-line: oklch(0.36 0.04 210); + --trace-card-border: oklch(0.27 0.02 235); + --trace-dot: oklch(0.45 0.06 195); + --trace-tab-active: oklch(0.28 0.03 215); + --trace-tab-hover: oklch(0.32 0.03 220); + --trace-tag: oklch(0.28 0.03 215); + --trace-chain: oklch(0.6 0.17 255); + --trace-llm: oklch(0.64 0.13 215); + --trace-tool: oklch(0.68 0.14 165); + --trace-glyph: oklch(0.99 0 0); + --trace-human: oklch(0.56 0.15 260); + --trace-human-glyph: oklch(0.95 0.04 210); + --trace-turn: oklch(0.29 0.05 210); + --trace-turn-border: oklch(0.5 0.09 200); + --trace-ok: oklch(0.35 0.07 160); + --trace-ok-glyph: oklch(0.82 0.15 155); + --trace-called: oklch(0.32 0.06 190); + --trace-called-text: oklch(0.88 0.08 185); + --trace-shadow-md: 0 4px 6px -1px #00000080, 0 2px 4px -1px #00000066; + --trace-shadow-xs: 0 1px 2px 0 #0000004d; } @theme inline { + --color-trace-text: var(--trace-text); + --color-trace-text-2: var(--trace-text-2); + --color-trace-text-secondary: var(--trace-text-secondary); + --color-trace-duration: var(--trace-duration); + --color-trace-key: var(--trace-key); + --color-trace-placeholder: var(--trace-placeholder); + --color-trace-surface: var(--trace-surface); + --color-trace-chip: var(--trace-chip); + --color-trace-row-hover: var(--trace-row-hover); + --color-trace-row-selected: var(--trace-row-selected); + --color-trace-brand: var(--trace-brand); + --color-trace-border: var(--trace-border); + --color-trace-line: var(--trace-line); + --color-trace-card-border: var(--trace-card-border); + --color-trace-dot: var(--trace-dot); + --color-trace-tab-active: var(--trace-tab-active); + --color-trace-tab-hover: var(--trace-tab-hover); + --color-trace-tag: var(--trace-tag); + --color-trace-chain: var(--trace-chain); + --color-trace-llm: var(--trace-llm); + --color-trace-tool: var(--trace-tool); + --color-trace-glyph: var(--trace-glyph); + --color-trace-human: var(--trace-human); + --color-trace-human-glyph: var(--trace-human-glyph); + --color-trace-turn: var(--trace-turn); + --color-trace-turn-border: var(--trace-turn-border); + --color-trace-ok: var(--trace-ok); + --color-trace-ok-glyph: var(--trace-ok-glyph); + --color-trace-called: var(--trace-called); + --color-trace-called-text: var(--trace-called-text); + --shadow-trace-md: var(--trace-shadow-md); + --shadow-trace-xs: var(--trace-shadow-xs); --radius-sm: calc(var(--radius) - 4px); --radius-md: calc(var(--radius) - 2px); --radius-lg: var(--radius); diff --git a/ui/litellm-dashboard/src/components/view_logs/TraceView/AgentTracesSection.test.tsx b/ui/litellm-dashboard/src/components/view_logs/TraceView/AgentTracesSection.test.tsx index 126542e848f..666859b1cc5 100644 --- a/ui/litellm-dashboard/src/components/view_logs/TraceView/AgentTracesSection.test.tsx +++ b/ui/litellm-dashboard/src/components/view_logs/TraceView/AgentTracesSection.test.tsx @@ -163,17 +163,51 @@ describe("AgentTracesSection", () => { expect(failed.length + ok.length).toBe(runs.length); }); - it("opens the run in place and goes back to the list", async () => { + it("opens a run in a side drawer over the list and swaps runs without closing it", async () => { vi.mocked(agentTraceListCall).mockResolvedValue(traceList as TracePage); renderSection(); const rows = await screen.findAllByTestId("agent-trace-row"); fireEvent.click(rows[0]); - expect(screen.getByTestId("run-view")).toHaveTextContent(`run ${runs[0].trace_id}`); - expect(screen.queryByTestId("runs-table")).not.toBeInTheDocument(); - - fireEvent.click(screen.getByText("back")); + const drawer = screen.getByRole("complementary", { name: "Trace details" }); + expect(within(drawer).getByTestId("run-view")).toHaveTextContent(`run ${runs[0].trace_id}`); expect(screen.getByTestId("runs-table")).toBeInTheDocument(); + expect(rows[0]).toHaveAttribute("aria-selected", "true"); + + fireEvent.click(rows[1]); + expect(screen.getByRole("complementary", { name: "Trace details" })).toBe(drawer); + expect(within(drawer).getByTestId("run-view")).toHaveTextContent(`run ${runs[1].trace_id}`); + expect(rows[1]).toHaveAttribute("aria-selected", "true"); + expect(rows[0]).toHaveAttribute("aria-selected", "false"); + }); + + it("closes the drawer when the open row is clicked again or Escape is pressed", async () => { + vi.mocked(agentTraceListCall).mockResolvedValue(traceList as TracePage); + renderSection(); + const rows = await screen.findAllByTestId("agent-trace-row"); + + fireEvent.click(rows[0]); + fireEvent.click(rows[0]); + expect(rows[0]).toHaveAttribute("aria-selected", "false"); + + fireEvent.click(rows[1]); + fireEvent.keyDown(window, { key: "Escape" }); + expect(rows[1]).toHaveAttribute("aria-selected", "false"); + }); + + it("moves to the next and previous run with j / k and the header arrows", async () => { + vi.mocked(agentTraceListCall).mockResolvedValue(traceList as TracePage); + renderSection(); + const rows = await screen.findAllByTestId("agent-trace-row"); + + fireEvent.click(rows[0]); + fireEvent.keyDown(window, { key: "j" }); + expect(screen.getByTestId("run-view")).toHaveTextContent(`run ${runs[1].trace_id}`); + fireEvent.keyDown(window, { key: "k" }); + expect(screen.getByTestId("run-view")).toHaveTextContent(`run ${runs[0].trace_id}`); + expect(screen.getByRole("button", { name: "Previous trace (K)" })).toBeDisabled(); + fireEvent.click(screen.getByRole("button", { name: "Next trace (J)" })); + expect(screen.getByTestId("run-view")).toHaveTextContent(`run ${runs[1].trace_id}`); }); it("plots every loaded run on the timeline", async () => { diff --git a/ui/litellm-dashboard/src/components/view_logs/TraceView/AgentTracesSection.tsx b/ui/litellm-dashboard/src/components/view_logs/TraceView/AgentTracesSection.tsx index 61e4c737b24..8f95573d0dd 100644 --- a/ui/litellm-dashboard/src/components/view_logs/TraceView/AgentTracesSection.tsx +++ b/ui/litellm-dashboard/src/components/view_logs/TraceView/AgentTracesSection.tsx @@ -1,11 +1,14 @@ "use client"; +import { Plug } from "lucide-react"; import moment from "moment"; import { useMemo, useState } from "react"; +import { Button } from "@/components/ui/button"; + import { AgentTracesTable } from "./AgentTracesTable"; +import { RunDrawer } from "./RunDrawer"; import { ALL_SERVICES, RunsToolbar, type RunStatusFilter } from "./RunsToolbar"; -import { RunView } from "./TraceDrawer"; import type { TraceSummary } from "./traceTypes"; import { previewText } from "./traceUtils"; import { TimeRangeControls } from "./TimeRangeControls"; @@ -31,6 +34,8 @@ export function filterRuns( }); } +const runKey = (run: TraceSummary): string => run.trace_ref || run.trace_id; + const filterByWindow = (runs: TraceSummary[], range: TimeWindow): TraceSummary[] => runs.filter((run) => { const t = moment(run.start_time).valueOf(); @@ -120,19 +125,12 @@ export function AgentTracesSection({ ); } - if (openTrace !== null) { - return ( - openRun(null)} - /> - ); - } + const toggleRun = (trace: TraceSummary | null) => + openRun(trace !== null && openTrace !== null && runKey(trace) === runKey(openTrace) ? null : trace); return (
+ - + {timeControls && (