feat(ui): session-level cache observability in request logs

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
yassin 2026-08-27 01:00:20 +00:00
parent 77765fd302
commit 1cd1b9dabe
12 changed files with 240 additions and 2 deletions

View file

@ -152,6 +152,7 @@ class _SessionSpendRow(TypedDict):
session_total_spend: float
mcp_tool_call_count: int
mcp_tool_call_spend: float
session_cache_hit_count: ReadOnly[int]
class _SpendSumAggregate(TypedDict, total=False):
@ -2239,6 +2240,10 @@ async def ui_view_spend_logs(
status_filter: str | None = fastapi.Query(
default=None, description="Filter logs by status (e.g., success, failure)"
),
cache_hit: str | None = fastapi.Query(
default=None,
description="Filter logs by response cache result: 'true' (served from cache) or 'false' (cache miss)",
),
model: str | None = fastapi.Query(default=None, description="Filter logs by model"),
model_id: str | None = fastapi.Query(
default=None,
@ -2311,6 +2316,13 @@ async def ui_view_spend_logs(
param="sort_order",
code=status.HTTP_400_BAD_REQUEST,
)
if cache_hit is not None and cache_hit.lower() not in {"true", "false"}:
raise ProxyException(
message=f"Invalid cache_hit: {cache_hit}. Must be one of: true, false",
type="bad_request",
param="cache_hit",
code=status.HTTP_400_BAD_REQUEST,
)
try:
is_admin_view: Final = _is_admin_view_safe(user_api_key_dict=user_api_key_dict)
@ -2542,6 +2554,12 @@ async def ui_view_spend_logs(
sql_params.append(f"%{like_escaped_session_id}%")
p += 1
if cache_hit is not None:
if cache_hit.lower() == "true":
sql_conditions.append("LOWER(cache_hit) = 'true'")
else:
sql_conditions.append("(cache_hit IS NULL OR LOWER(cache_hit) <> 'true')")
# Status filter
if status_filter is not None:
if status_filter == "success":
@ -4093,7 +4111,8 @@ async def _build_ui_spend_logs_response(
)::int AS mcp_tool_call_count,
COALESCE(SUM(spend) FILTER (
WHERE call_type IN ('call_mcp_tool', 'list_mcp_tools')
), 0)::double precision AS mcp_tool_call_spend
), 0)::double precision AS mcp_tool_call_spend,
COUNT(*) FILTER (WHERE LOWER(cache_hit) = 'true')::int AS session_cache_hit_count
FROM "LiteLLM_SpendLogs"
WHERE session_id = ANY($1::text[])
AND api_key = ANY($2::text[])
@ -4107,6 +4126,7 @@ async def _build_ui_spend_logs_response(
"session_total_spend": float(row.get("session_total_spend") or 0.0),
"mcp_tool_call_count": int(row.get("mcp_tool_call_count") or 0),
"mcp_tool_call_spend": float(row.get("mcp_tool_call_spend") or 0.0),
"session_cache_hit_count": int(row.get("session_cache_hit_count") or 0),
}
for row in rows
if row.get("session_id")
@ -4129,6 +4149,7 @@ async def _build_ui_spend_logs_response(
if session_stats["mcp_tool_call_count"]:
row_dict["mcp_tool_call_count"] = session_stats["mcp_tool_call_count"]
row_dict["mcp_tool_call_spend"] = session_stats["mcp_tool_call_spend"]
row_dict["session_cache_hit_count"] = session_stats["session_cache_hit_count"]
enriched.append(row_dict)
response_data: list = enriched
else:

View file

@ -104,6 +104,10 @@ def _reconstruct_ui_where_from_sql(sql_query, params):
where["OR"] = where.get("OR", []) + [{"multi_team": True}]
elif "status = 'success'" in cond:
where["OR"] = where.get("OR", []) + [{"status": "success"}]
elif cond == "LOWER(cache_hit) = 'true'":
where["cache_hit"] = {"is_hit": True}
elif "cache_hit IS NULL" in cond:
where["cache_hit"] = {"is_hit": False}
elif sess:
where["session_id"] = {"contains": str(params[int(sess.group(1)) - 1]).strip("%")}
elif status:
@ -2302,6 +2306,100 @@ async def test_ui_view_spend_logs_with_status(client, monkeypatch):
app.dependency_overrides.pop(ps.user_api_key_auth, None)
@pytest.mark.asyncio
async def test_ui_view_spend_logs_with_cache_hit(client, monkeypatch):
mock_spend_logs = [
{
"id": "log1",
"request_id": "req1",
"api_key": "sk-test-key",
"user": "test_user_1",
"team_id": "team1",
"spend": 0.05,
"startTime": datetime.datetime.now(timezone.utc).isoformat(),
"model": "gpt-3.5-turbo",
"cache_hit": "True",
},
{
"id": "log2",
"request_id": "req2",
"api_key": "sk-test-key",
"user": "test_user_2",
"team_id": "team1",
"spend": 0.10,
"startTime": datetime.datetime.now(timezone.utc).isoformat(),
"model": "gpt-4",
"cache_hit": "False",
},
]
def filter_by_cache_hit(where):
cache_cond = where.get("cache_hit")
if cache_cond is None:
return mock_spend_logs
if cache_cond["is_hit"]:
return [log for log in mock_spend_logs if str(log.get("cache_hit", "")).lower() == "true"]
return [log for log in mock_spend_logs if str(log.get("cache_hit", "")).lower() != "true"]
monkeypatch.setattr(
"litellm.proxy.proxy_server.prisma_client",
make_ui_spend_logs_mock_prisma(mock_spend_logs, filter_by_cache_hit),
)
start_date, end_date = _default_date_range()
app.dependency_overrides[ps.user_api_key_auth] = lambda: UserAPIKeyAuth(
user_role=LitellmUserRoles.PROXY_ADMIN
)
try:
response = client.get(
"/spend/logs/ui",
params={
"cache_hit": "true",
"start_date": start_date,
"end_date": end_date,
},
headers={"Authorization": "Bearer sk-test"},
)
assert response.status_code == 200
data = response.json()
assert data["total"] == 1
assert len(data["data"]) == 1
assert data["data"][0]["request_id"] == "req1"
response = client.get(
"/spend/logs/ui",
params={
"cache_hit": "false",
"start_date": start_date,
"end_date": end_date,
},
headers={"Authorization": "Bearer sk-test"},
)
assert response.status_code == 200
data = response.json()
assert data["total"] == 1
assert len(data["data"]) == 1
assert data["data"][0]["request_id"] == "req2"
response = client.get(
"/spend/logs/ui",
params={
"cache_hit": "maybe",
"start_date": start_date,
"end_date": end_date,
},
headers={"Authorization": "Bearer sk-test"},
)
assert response.status_code == 400
assert "cache_hit" in response.text
finally:
app.dependency_overrides.pop(ps.user_api_key_auth, None)
@pytest.mark.asyncio
async def test_ui_view_spend_logs_with_model(client, monkeypatch):
mock_spend_logs = [
@ -3749,6 +3847,62 @@ async def test_build_ui_spend_logs_response_sums_multi_round_session_spend():
assert call_args[2] == [api_key]
@pytest.mark.asyncio
async def test_build_ui_spend_logs_response_session_cache_hit_count():
"""
Each row of a session must carry session_cache_hit_count aggregated across
the whole session so the UI can show how many requests in the session were
served from the response cache.
"""
from litellm.proxy.spend_tracking.spend_management_endpoints import (
_build_ui_spend_logs_response,
)
session_id = "sess-cache-hits"
api_key = "hashed-key-xyz"
dict_rows = [
{"request_id": "req-1", "session_id": session_id, "call_type": "completion", "api_key": api_key},
{"request_id": "req-2", "session_id": session_id, "call_type": "completion", "api_key": api_key},
{"request_id": "req-3", "session_id": None, "call_type": "completion", "api_key": api_key},
]
mock_prisma = MagicMock()
mock_prisma.db.litellm_spendlogs.group_by = AsyncMock(
return_value=[{"session_id": session_id, "_count": {"session_id": 2}}]
)
mock_prisma.db.query_raw = AsyncMock(
return_value=[
{
"session_id": session_id,
"session_total_spend": 0.05,
"mcp_tool_call_count": 0,
"mcp_tool_call_spend": 0.0,
"session_cache_hit_count": 2,
}
]
)
result = await _build_ui_spend_logs_response(
prisma_client=mock_prisma,
data=dict_rows,
total_records=3,
page=1,
page_size=50,
total_pages=1,
enrich_session_counts=True,
)
rows = result["data"]
assert rows[0]["session_cache_hit_count"] == 2
assert rows[1]["session_cache_hit_count"] == 2
assert "session_cache_hit_count" not in rows[2]
# The aggregate SQL must actually compute the cache-hit count.
_, call_args, _ = mock_prisma.db.query_raw.mock_calls[0]
assert "session_cache_hit_count" in call_args[0]
assert "LOWER(cache_hit) = 'true'" in call_args[0]
# ---------------------------------------------------------------------------
# Tests for /spend/logs team-member permission
# ---------------------------------------------------------------------------

View file

@ -2002,6 +2002,8 @@ interface UiSpendLogsParams {
user_id?: string;
end_user?: string;
status_filter?: string;
/** Filter by response cache result: "true" (cache hit) or "false" (cache miss) */
cache_hit?: string;
/** Filter by model name (e.g. "gpt-4") */
model?: string;
/** Filter by model ID (litellm model deployment id) */

View file

@ -359,6 +359,19 @@ describe("LogDetailContent", () => {
expect(screen.queryByText("Response Cache")).not.toBeInTheDocument();
});
it("should display the Cache Key next to the Response Cache result", () => {
render(<LogDetailContent logEntry={createLogEntry({ cache_hit: "True", cache_key: "abc123cachekey" })} />);
expect(screen.getByText("Cache Key")).toBeInTheDocument();
expect(screen.getByText("abc123cachekey")).toBeInTheDocument();
});
it("should hide the Cache Key row when caching is off", () => {
render(<LogDetailContent logEntry={createLogEntry({ cache_hit: "False", cache_key: "Cache OFF" })} />);
expect(screen.queryByText("Cache Key")).not.toBeInTheDocument();
});
it("should display LiteLLM Overhead when litellm_overhead_time_ms is in metadata", () => {
render(
<LogDetailContent

View file

@ -347,6 +347,8 @@ function getUncachedInputTextTokens(metadata: Record<string, any>): number | und
const RESPONSE_CACHE_TOOLTIP =
"Whether this request was served from LiteLLM's response cache (e.g. Redis / in-memory), skipping the LLM provider call entirely. This is separate from provider prompt caching; a Miss here does not mean prompt caching failed.";
const RESPONSE_CACHE_DOCS_URL = "https://docs.litellm.ai/docs/proxy/caching";
const CACHE_KEY_TOOLTIP =
"The key LiteLLM computed for this request in the response cache. Requests with the same cache key share a cached response; a different key means the request content did not match any cached entry.";
const PROMPT_CACHE_DOCS_URL = "https://docs.litellm.ai/docs/completion/prompt_caching";
function MetricLabel({ label, tooltip, docsUrl }: { label: string; tooltip: string; docsUrl: string }) {
@ -436,6 +438,13 @@ function MetricsSection({ logEntry, metadata }: { logEntry: LogEntry; metadata:
</Badge>
</DescriptionItem>
)}
{showResponseCache && logEntry.cache_key && logEntry.cache_key !== "Cache OFF" && (
<DescriptionItem
label={<MetricLabel label="Cache Key" tooltip={CACHE_KEY_TOOLTIP} docsUrl={RESPONSE_CACHE_DOCS_URL} />}
>
<TruncatedValue value={logEntry.cache_key} />
</DescriptionItem>
)}
{promptCacheReadTokens > 0 && (
<DescriptionItem
label={

View file

@ -278,6 +278,7 @@ export function LogDetailsDrawer({
).length;
const agentCount = sessionLogs.filter((row) => AGENT_CALL_TYPES.includes(row.call_type)).length;
const mcpCount = sessionLogs.filter((row) => MCP_CALL_TYPES.includes(row.call_type)).length;
const cacheHitCount = sessionLogs.filter((row) => String(row.cache_hit ?? "").toLowerCase() === "true").length;
const logsForList = isSessionMode ? sessionLogs : currentLog ? [currentLog] : [];
const leftPanelId = isSessionMode ? sessionId || "" : currentLog?.request_id || "";
const leftPanelDisplayId = leftPanelId.length > 14 ? `${leftPanelId.slice(0, 11)}...` : leftPanelId;
@ -383,7 +384,8 @@ export function LogDetailsDrawer({
{isSessionMode && (
<>
<span className="mx-1.5">·</span>
{sessionDurationSeconds}s
{sessionDurationSeconds}s<span className="mx-1.5">·</span>
{cacheHitCount}/{logsForList.length} cached
</>
)}
</div>

View file

@ -31,6 +31,12 @@ const STATUS_FILTER_ITEMS = [
{ value: "success", label: "Success" },
{ value: "failure", label: "Failure" },
] as const;
const CACHE_HIT_FILTER_ITEMS = [
{ value: ALL_VALUE, label: "All Requests" },
{ value: "true", label: "Cache Hit" },
{ value: "false", label: "Cache Miss" },
] as const;
const PAGE_SIZE = 50;
const asString = (value: unknown): string => (typeof value === "string" ? value : "");
@ -364,6 +370,27 @@ export function RequestLogsFilters({ get, set, teams, logsWindow }: RequestLogsF
/>
</DataTableFilterField>
<DataTableFilterField label="Cache Hit">
<Select
items={CACHE_HIT_FILTER_ITEMS}
value={valueOf(LOG_FILTER_IDS.CACHE_HIT) === "" ? ALL_VALUE : valueOf(LOG_FILTER_IDS.CACHE_HIT)}
onValueChange={(next) =>
set(LOG_FILTER_IDS.CACHE_HIT, next === null || next === ALL_VALUE ? undefined : next)
}
>
<SelectTrigger className="w-full">
<SelectValue placeholder="All Requests" />
</SelectTrigger>
<SelectContent>
{CACHE_HIT_FILTER_ITEMS.map((item) => (
<SelectItem key={item.value} value={item.value}>
{item.label}
</SelectItem>
))}
</SelectContent>
</Select>
</DataTableFilterField>
<DataTableFilterField label="Session ID">
<Input
value={valueOf(LOG_FILTER_IDS.SESSION_ID)}

View file

@ -90,6 +90,7 @@ export const getRequestLogsTableColumns = ({
sessionLlmCount > 0 && `${sessionLlmCount} LLM`,
sessionAgentCount > 0 && `${sessionAgentCount} Agent`,
sessionMcpCount > 0 && `${sessionMcpCount} MCP`,
log.session_cache_hit_count != null && `${log.session_cache_hit_count} cache hit`,
].filter(Boolean);
return <CellTooltip content={tooltipParts.join(" • ")} trigger={sessionTypeBadge} />;
},

View file

@ -42,6 +42,7 @@ export type LogEntry = {
request_duration_ms?: number;
session_total_count?: number;
session_total_spend?: number;
session_cache_hit_count?: number;
mcp_tool_call_count?: number;
mcp_tool_call_spend?: number;
session_llm_count?: number;

View file

@ -82,6 +82,7 @@ describe("useLogFilterLogic", () => {
{ id: LOG_FILTER_IDS.SESSION_ID, value: "sess-1", param: "session_id" },
{ id: LOG_FILTER_IDS.END_USER, value: "end-user-1", param: "end_user" },
{ id: LOG_FILTER_IDS.STATUS, value: "failure", param: "status_filter" },
{ id: LOG_FILTER_IDS.CACHE_HIT, value: "true", param: "cache_hit" },
{ id: LOG_FILTER_IDS.MODEL_ID, value: "model-uuid-1", param: "model_id" },
{ id: LOG_FILTER_IDS.PUBLIC_MODEL_OR_SEARCH_TOOL, value: "gpt-4o", param: "model" },
{ id: LOG_FILTER_IDS.KEY_ALIAS, value: "alias-1", param: "key_alias" },

View file

@ -26,6 +26,7 @@ export const LOG_FILTER_IDS = {
ERROR_MESSAGE: "error_message",
KEY_HASH: "key_hash",
SESSION_ID: "session_id",
CACHE_HIT: "cache_hit",
MODEL_ID: "model_id",
PUBLIC_MODEL_OR_SEARCH_TOOL: "model",
REQUEST_ID: "request_id",
@ -42,6 +43,7 @@ export const LOG_FILTER_LABELS: Record<string, string> = {
[LOG_FILTER_IDS.ERROR_MESSAGE]: "Error Message",
[LOG_FILTER_IDS.KEY_HASH]: "Key Hash",
[LOG_FILTER_IDS.SESSION_ID]: "Session ID",
[LOG_FILTER_IDS.CACHE_HIT]: "Cache Hit",
[LOG_FILTER_IDS.MODEL_ID]: "Model",
[LOG_FILTER_IDS.PUBLIC_MODEL_OR_SEARCH_TOOL]: "Public model / search tool",
};
@ -167,6 +169,7 @@ export function useLogFilterLogic({
user_id: userIdFilter,
end_user: getFilterValue(columnFilters, LOG_FILTER_IDS.END_USER),
status_filter: getFilterValue(columnFilters, LOG_FILTER_IDS.STATUS),
cache_hit: getFilterValue(columnFilters, LOG_FILTER_IDS.CACHE_HIT),
model_id: getFilterValue(columnFilters, LOG_FILTER_IDS.MODEL_ID),
model: getFilterValue(columnFilters, LOG_FILTER_IDS.PUBLIC_MODEL_OR_SEARCH_TOOL),
key_alias: getFilterValue(columnFilters, LOG_FILTER_IDS.KEY_ALIAS),

View file

@ -53273,6 +53273,8 @@ export interface operations {
page_size?: number;
/** @description Filter logs by status (e.g., success, failure) */
status_filter?: string | null;
/** @description Filter logs by response cache result: 'true' (served from cache) or 'false' (cache miss) */
cache_hit?: string | null;
/** @description Filter logs by model */
model?: string | null;
/** @description Filter logs by model ID (litellm model deployment id) */
@ -53381,6 +53383,8 @@ export interface operations {
page_size?: number;
/** @description Filter logs by status (e.g., success, failure) */
status_filter?: string | null;
/** @description Filter logs by response cache result: 'true' (served from cache) or 'false' (cache miss) */
cache_hit?: string | null;
/** @description Filter logs by model */
model?: string | null;
/** @description Filter logs by model ID (litellm model deployment id) */