mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
feat(ui): session-level cache observability in request logs
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
77765fd302
commit
1cd1b9dabe
12 changed files with 240 additions and 2 deletions
|
|
@ -152,6 +152,7 @@ class _SessionSpendRow(TypedDict):
|
|||
session_total_spend: float
|
||||
mcp_tool_call_count: int
|
||||
mcp_tool_call_spend: float
|
||||
session_cache_hit_count: ReadOnly[int]
|
||||
|
||||
|
||||
class _SpendSumAggregate(TypedDict, total=False):
|
||||
|
|
@ -2239,6 +2240,10 @@ async def ui_view_spend_logs(
|
|||
status_filter: str | None = fastapi.Query(
|
||||
default=None, description="Filter logs by status (e.g., success, failure)"
|
||||
),
|
||||
cache_hit: str | None = fastapi.Query(
|
||||
default=None,
|
||||
description="Filter logs by response cache result: 'true' (served from cache) or 'false' (cache miss)",
|
||||
),
|
||||
model: str | None = fastapi.Query(default=None, description="Filter logs by model"),
|
||||
model_id: str | None = fastapi.Query(
|
||||
default=None,
|
||||
|
|
@ -2311,6 +2316,13 @@ async def ui_view_spend_logs(
|
|||
param="sort_order",
|
||||
code=status.HTTP_400_BAD_REQUEST,
|
||||
)
|
||||
if cache_hit is not None and cache_hit.lower() not in {"true", "false"}:
|
||||
raise ProxyException(
|
||||
message=f"Invalid cache_hit: {cache_hit}. Must be one of: true, false",
|
||||
type="bad_request",
|
||||
param="cache_hit",
|
||||
code=status.HTTP_400_BAD_REQUEST,
|
||||
)
|
||||
|
||||
try:
|
||||
is_admin_view: Final = _is_admin_view_safe(user_api_key_dict=user_api_key_dict)
|
||||
|
|
@ -2542,6 +2554,12 @@ async def ui_view_spend_logs(
|
|||
sql_params.append(f"%{like_escaped_session_id}%")
|
||||
p += 1
|
||||
|
||||
if cache_hit is not None:
|
||||
if cache_hit.lower() == "true":
|
||||
sql_conditions.append("LOWER(cache_hit) = 'true'")
|
||||
else:
|
||||
sql_conditions.append("(cache_hit IS NULL OR LOWER(cache_hit) <> 'true')")
|
||||
|
||||
# Status filter
|
||||
if status_filter is not None:
|
||||
if status_filter == "success":
|
||||
|
|
@ -4093,7 +4111,8 @@ async def _build_ui_spend_logs_response(
|
|||
)::int AS mcp_tool_call_count,
|
||||
COALESCE(SUM(spend) FILTER (
|
||||
WHERE call_type IN ('call_mcp_tool', 'list_mcp_tools')
|
||||
), 0)::double precision AS mcp_tool_call_spend
|
||||
), 0)::double precision AS mcp_tool_call_spend,
|
||||
COUNT(*) FILTER (WHERE LOWER(cache_hit) = 'true')::int AS session_cache_hit_count
|
||||
FROM "LiteLLM_SpendLogs"
|
||||
WHERE session_id = ANY($1::text[])
|
||||
AND api_key = ANY($2::text[])
|
||||
|
|
@ -4107,6 +4126,7 @@ async def _build_ui_spend_logs_response(
|
|||
"session_total_spend": float(row.get("session_total_spend") or 0.0),
|
||||
"mcp_tool_call_count": int(row.get("mcp_tool_call_count") or 0),
|
||||
"mcp_tool_call_spend": float(row.get("mcp_tool_call_spend") or 0.0),
|
||||
"session_cache_hit_count": int(row.get("session_cache_hit_count") or 0),
|
||||
}
|
||||
for row in rows
|
||||
if row.get("session_id")
|
||||
|
|
@ -4129,6 +4149,7 @@ async def _build_ui_spend_logs_response(
|
|||
if session_stats["mcp_tool_call_count"]:
|
||||
row_dict["mcp_tool_call_count"] = session_stats["mcp_tool_call_count"]
|
||||
row_dict["mcp_tool_call_spend"] = session_stats["mcp_tool_call_spend"]
|
||||
row_dict["session_cache_hit_count"] = session_stats["session_cache_hit_count"]
|
||||
enriched.append(row_dict)
|
||||
response_data: list = enriched
|
||||
else:
|
||||
|
|
|
|||
|
|
@ -104,6 +104,10 @@ def _reconstruct_ui_where_from_sql(sql_query, params):
|
|||
where["OR"] = where.get("OR", []) + [{"multi_team": True}]
|
||||
elif "status = 'success'" in cond:
|
||||
where["OR"] = where.get("OR", []) + [{"status": "success"}]
|
||||
elif cond == "LOWER(cache_hit) = 'true'":
|
||||
where["cache_hit"] = {"is_hit": True}
|
||||
elif "cache_hit IS NULL" in cond:
|
||||
where["cache_hit"] = {"is_hit": False}
|
||||
elif sess:
|
||||
where["session_id"] = {"contains": str(params[int(sess.group(1)) - 1]).strip("%")}
|
||||
elif status:
|
||||
|
|
@ -2302,6 +2306,100 @@ async def test_ui_view_spend_logs_with_status(client, monkeypatch):
|
|||
app.dependency_overrides.pop(ps.user_api_key_auth, None)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_ui_view_spend_logs_with_cache_hit(client, monkeypatch):
|
||||
mock_spend_logs = [
|
||||
{
|
||||
"id": "log1",
|
||||
"request_id": "req1",
|
||||
"api_key": "sk-test-key",
|
||||
"user": "test_user_1",
|
||||
"team_id": "team1",
|
||||
"spend": 0.05,
|
||||
"startTime": datetime.datetime.now(timezone.utc).isoformat(),
|
||||
"model": "gpt-3.5-turbo",
|
||||
"cache_hit": "True",
|
||||
},
|
||||
{
|
||||
"id": "log2",
|
||||
"request_id": "req2",
|
||||
"api_key": "sk-test-key",
|
||||
"user": "test_user_2",
|
||||
"team_id": "team1",
|
||||
"spend": 0.10,
|
||||
"startTime": datetime.datetime.now(timezone.utc).isoformat(),
|
||||
"model": "gpt-4",
|
||||
"cache_hit": "False",
|
||||
},
|
||||
]
|
||||
|
||||
def filter_by_cache_hit(where):
|
||||
cache_cond = where.get("cache_hit")
|
||||
if cache_cond is None:
|
||||
return mock_spend_logs
|
||||
if cache_cond["is_hit"]:
|
||||
return [log for log in mock_spend_logs if str(log.get("cache_hit", "")).lower() == "true"]
|
||||
return [log for log in mock_spend_logs if str(log.get("cache_hit", "")).lower() != "true"]
|
||||
|
||||
monkeypatch.setattr(
|
||||
"litellm.proxy.proxy_server.prisma_client",
|
||||
make_ui_spend_logs_mock_prisma(mock_spend_logs, filter_by_cache_hit),
|
||||
)
|
||||
|
||||
start_date, end_date = _default_date_range()
|
||||
|
||||
app.dependency_overrides[ps.user_api_key_auth] = lambda: UserAPIKeyAuth(
|
||||
user_role=LitellmUserRoles.PROXY_ADMIN
|
||||
)
|
||||
try:
|
||||
response = client.get(
|
||||
"/spend/logs/ui",
|
||||
params={
|
||||
"cache_hit": "true",
|
||||
"start_date": start_date,
|
||||
"end_date": end_date,
|
||||
},
|
||||
headers={"Authorization": "Bearer sk-test"},
|
||||
)
|
||||
|
||||
assert response.status_code == 200
|
||||
data = response.json()
|
||||
assert data["total"] == 1
|
||||
assert len(data["data"]) == 1
|
||||
assert data["data"][0]["request_id"] == "req1"
|
||||
|
||||
response = client.get(
|
||||
"/spend/logs/ui",
|
||||
params={
|
||||
"cache_hit": "false",
|
||||
"start_date": start_date,
|
||||
"end_date": end_date,
|
||||
},
|
||||
headers={"Authorization": "Bearer sk-test"},
|
||||
)
|
||||
|
||||
assert response.status_code == 200
|
||||
data = response.json()
|
||||
assert data["total"] == 1
|
||||
assert len(data["data"]) == 1
|
||||
assert data["data"][0]["request_id"] == "req2"
|
||||
|
||||
response = client.get(
|
||||
"/spend/logs/ui",
|
||||
params={
|
||||
"cache_hit": "maybe",
|
||||
"start_date": start_date,
|
||||
"end_date": end_date,
|
||||
},
|
||||
headers={"Authorization": "Bearer sk-test"},
|
||||
)
|
||||
|
||||
assert response.status_code == 400
|
||||
assert "cache_hit" in response.text
|
||||
finally:
|
||||
app.dependency_overrides.pop(ps.user_api_key_auth, None)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_ui_view_spend_logs_with_model(client, monkeypatch):
|
||||
mock_spend_logs = [
|
||||
|
|
@ -3749,6 +3847,62 @@ async def test_build_ui_spend_logs_response_sums_multi_round_session_spend():
|
|||
assert call_args[2] == [api_key]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_build_ui_spend_logs_response_session_cache_hit_count():
|
||||
"""
|
||||
Each row of a session must carry session_cache_hit_count aggregated across
|
||||
the whole session so the UI can show how many requests in the session were
|
||||
served from the response cache.
|
||||
"""
|
||||
from litellm.proxy.spend_tracking.spend_management_endpoints import (
|
||||
_build_ui_spend_logs_response,
|
||||
)
|
||||
|
||||
session_id = "sess-cache-hits"
|
||||
api_key = "hashed-key-xyz"
|
||||
dict_rows = [
|
||||
{"request_id": "req-1", "session_id": session_id, "call_type": "completion", "api_key": api_key},
|
||||
{"request_id": "req-2", "session_id": session_id, "call_type": "completion", "api_key": api_key},
|
||||
{"request_id": "req-3", "session_id": None, "call_type": "completion", "api_key": api_key},
|
||||
]
|
||||
|
||||
mock_prisma = MagicMock()
|
||||
mock_prisma.db.litellm_spendlogs.group_by = AsyncMock(
|
||||
return_value=[{"session_id": session_id, "_count": {"session_id": 2}}]
|
||||
)
|
||||
mock_prisma.db.query_raw = AsyncMock(
|
||||
return_value=[
|
||||
{
|
||||
"session_id": session_id,
|
||||
"session_total_spend": 0.05,
|
||||
"mcp_tool_call_count": 0,
|
||||
"mcp_tool_call_spend": 0.0,
|
||||
"session_cache_hit_count": 2,
|
||||
}
|
||||
]
|
||||
)
|
||||
|
||||
result = await _build_ui_spend_logs_response(
|
||||
prisma_client=mock_prisma,
|
||||
data=dict_rows,
|
||||
total_records=3,
|
||||
page=1,
|
||||
page_size=50,
|
||||
total_pages=1,
|
||||
enrich_session_counts=True,
|
||||
)
|
||||
|
||||
rows = result["data"]
|
||||
assert rows[0]["session_cache_hit_count"] == 2
|
||||
assert rows[1]["session_cache_hit_count"] == 2
|
||||
assert "session_cache_hit_count" not in rows[2]
|
||||
|
||||
# The aggregate SQL must actually compute the cache-hit count.
|
||||
_, call_args, _ = mock_prisma.db.query_raw.mock_calls[0]
|
||||
assert "session_cache_hit_count" in call_args[0]
|
||||
assert "LOWER(cache_hit) = 'true'" in call_args[0]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Tests for /spend/logs team-member permission
|
||||
# ---------------------------------------------------------------------------
|
||||
|
|
|
|||
|
|
@ -2002,6 +2002,8 @@ interface UiSpendLogsParams {
|
|||
user_id?: string;
|
||||
end_user?: string;
|
||||
status_filter?: string;
|
||||
/** Filter by response cache result: "true" (cache hit) or "false" (cache miss) */
|
||||
cache_hit?: string;
|
||||
/** Filter by model name (e.g. "gpt-4") */
|
||||
model?: string;
|
||||
/** Filter by model ID (litellm model deployment id) */
|
||||
|
|
|
|||
|
|
@ -359,6 +359,19 @@ describe("LogDetailContent", () => {
|
|||
expect(screen.queryByText("Response Cache")).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("should display the Cache Key next to the Response Cache result", () => {
|
||||
render(<LogDetailContent logEntry={createLogEntry({ cache_hit: "True", cache_key: "abc123cachekey" })} />);
|
||||
|
||||
expect(screen.getByText("Cache Key")).toBeInTheDocument();
|
||||
expect(screen.getByText("abc123cachekey")).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("should hide the Cache Key row when caching is off", () => {
|
||||
render(<LogDetailContent logEntry={createLogEntry({ cache_hit: "False", cache_key: "Cache OFF" })} />);
|
||||
|
||||
expect(screen.queryByText("Cache Key")).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("should display LiteLLM Overhead when litellm_overhead_time_ms is in metadata", () => {
|
||||
render(
|
||||
<LogDetailContent
|
||||
|
|
|
|||
|
|
@ -347,6 +347,8 @@ function getUncachedInputTextTokens(metadata: Record<string, any>): number | und
|
|||
const RESPONSE_CACHE_TOOLTIP =
|
||||
"Whether this request was served from LiteLLM's response cache (e.g. Redis / in-memory), skipping the LLM provider call entirely. This is separate from provider prompt caching; a Miss here does not mean prompt caching failed.";
|
||||
const RESPONSE_CACHE_DOCS_URL = "https://docs.litellm.ai/docs/proxy/caching";
|
||||
const CACHE_KEY_TOOLTIP =
|
||||
"The key LiteLLM computed for this request in the response cache. Requests with the same cache key share a cached response; a different key means the request content did not match any cached entry.";
|
||||
const PROMPT_CACHE_DOCS_URL = "https://docs.litellm.ai/docs/completion/prompt_caching";
|
||||
|
||||
function MetricLabel({ label, tooltip, docsUrl }: { label: string; tooltip: string; docsUrl: string }) {
|
||||
|
|
@ -436,6 +438,13 @@ function MetricsSection({ logEntry, metadata }: { logEntry: LogEntry; metadata:
|
|||
</Badge>
|
||||
</DescriptionItem>
|
||||
)}
|
||||
{showResponseCache && logEntry.cache_key && logEntry.cache_key !== "Cache OFF" && (
|
||||
<DescriptionItem
|
||||
label={<MetricLabel label="Cache Key" tooltip={CACHE_KEY_TOOLTIP} docsUrl={RESPONSE_CACHE_DOCS_URL} />}
|
||||
>
|
||||
<TruncatedValue value={logEntry.cache_key} />
|
||||
</DescriptionItem>
|
||||
)}
|
||||
{promptCacheReadTokens > 0 && (
|
||||
<DescriptionItem
|
||||
label={
|
||||
|
|
|
|||
|
|
@ -278,6 +278,7 @@ export function LogDetailsDrawer({
|
|||
).length;
|
||||
const agentCount = sessionLogs.filter((row) => AGENT_CALL_TYPES.includes(row.call_type)).length;
|
||||
const mcpCount = sessionLogs.filter((row) => MCP_CALL_TYPES.includes(row.call_type)).length;
|
||||
const cacheHitCount = sessionLogs.filter((row) => String(row.cache_hit ?? "").toLowerCase() === "true").length;
|
||||
const logsForList = isSessionMode ? sessionLogs : currentLog ? [currentLog] : [];
|
||||
const leftPanelId = isSessionMode ? sessionId || "" : currentLog?.request_id || "";
|
||||
const leftPanelDisplayId = leftPanelId.length > 14 ? `${leftPanelId.slice(0, 11)}...` : leftPanelId;
|
||||
|
|
@ -383,7 +384,8 @@ export function LogDetailsDrawer({
|
|||
{isSessionMode && (
|
||||
<>
|
||||
<span className="mx-1.5">·</span>
|
||||
{sessionDurationSeconds}s
|
||||
{sessionDurationSeconds}s<span className="mx-1.5">·</span>
|
||||
{cacheHitCount}/{logsForList.length} cached
|
||||
</>
|
||||
)}
|
||||
</div>
|
||||
|
|
|
|||
|
|
@ -31,6 +31,12 @@ const STATUS_FILTER_ITEMS = [
|
|||
{ value: "success", label: "Success" },
|
||||
{ value: "failure", label: "Failure" },
|
||||
] as const;
|
||||
|
||||
const CACHE_HIT_FILTER_ITEMS = [
|
||||
{ value: ALL_VALUE, label: "All Requests" },
|
||||
{ value: "true", label: "Cache Hit" },
|
||||
{ value: "false", label: "Cache Miss" },
|
||||
] as const;
|
||||
const PAGE_SIZE = 50;
|
||||
|
||||
const asString = (value: unknown): string => (typeof value === "string" ? value : "");
|
||||
|
|
@ -364,6 +370,27 @@ export function RequestLogsFilters({ get, set, teams, logsWindow }: RequestLogsF
|
|||
/>
|
||||
</DataTableFilterField>
|
||||
|
||||
<DataTableFilterField label="Cache Hit">
|
||||
<Select
|
||||
items={CACHE_HIT_FILTER_ITEMS}
|
||||
value={valueOf(LOG_FILTER_IDS.CACHE_HIT) === "" ? ALL_VALUE : valueOf(LOG_FILTER_IDS.CACHE_HIT)}
|
||||
onValueChange={(next) =>
|
||||
set(LOG_FILTER_IDS.CACHE_HIT, next === null || next === ALL_VALUE ? undefined : next)
|
||||
}
|
||||
>
|
||||
<SelectTrigger className="w-full">
|
||||
<SelectValue placeholder="All Requests" />
|
||||
</SelectTrigger>
|
||||
<SelectContent>
|
||||
{CACHE_HIT_FILTER_ITEMS.map((item) => (
|
||||
<SelectItem key={item.value} value={item.value}>
|
||||
{item.label}
|
||||
</SelectItem>
|
||||
))}
|
||||
</SelectContent>
|
||||
</Select>
|
||||
</DataTableFilterField>
|
||||
|
||||
<DataTableFilterField label="Session ID">
|
||||
<Input
|
||||
value={valueOf(LOG_FILTER_IDS.SESSION_ID)}
|
||||
|
|
|
|||
|
|
@ -90,6 +90,7 @@ export const getRequestLogsTableColumns = ({
|
|||
sessionLlmCount > 0 && `${sessionLlmCount} LLM`,
|
||||
sessionAgentCount > 0 && `${sessionAgentCount} Agent`,
|
||||
sessionMcpCount > 0 && `${sessionMcpCount} MCP`,
|
||||
log.session_cache_hit_count != null && `${log.session_cache_hit_count} cache hit`,
|
||||
].filter(Boolean);
|
||||
return <CellTooltip content={tooltipParts.join(" • ")} trigger={sessionTypeBadge} />;
|
||||
},
|
||||
|
|
|
|||
|
|
@ -42,6 +42,7 @@ export type LogEntry = {
|
|||
request_duration_ms?: number;
|
||||
session_total_count?: number;
|
||||
session_total_spend?: number;
|
||||
session_cache_hit_count?: number;
|
||||
mcp_tool_call_count?: number;
|
||||
mcp_tool_call_spend?: number;
|
||||
session_llm_count?: number;
|
||||
|
|
|
|||
|
|
@ -82,6 +82,7 @@ describe("useLogFilterLogic", () => {
|
|||
{ id: LOG_FILTER_IDS.SESSION_ID, value: "sess-1", param: "session_id" },
|
||||
{ id: LOG_FILTER_IDS.END_USER, value: "end-user-1", param: "end_user" },
|
||||
{ id: LOG_FILTER_IDS.STATUS, value: "failure", param: "status_filter" },
|
||||
{ id: LOG_FILTER_IDS.CACHE_HIT, value: "true", param: "cache_hit" },
|
||||
{ id: LOG_FILTER_IDS.MODEL_ID, value: "model-uuid-1", param: "model_id" },
|
||||
{ id: LOG_FILTER_IDS.PUBLIC_MODEL_OR_SEARCH_TOOL, value: "gpt-4o", param: "model" },
|
||||
{ id: LOG_FILTER_IDS.KEY_ALIAS, value: "alias-1", param: "key_alias" },
|
||||
|
|
|
|||
|
|
@ -26,6 +26,7 @@ export const LOG_FILTER_IDS = {
|
|||
ERROR_MESSAGE: "error_message",
|
||||
KEY_HASH: "key_hash",
|
||||
SESSION_ID: "session_id",
|
||||
CACHE_HIT: "cache_hit",
|
||||
MODEL_ID: "model_id",
|
||||
PUBLIC_MODEL_OR_SEARCH_TOOL: "model",
|
||||
REQUEST_ID: "request_id",
|
||||
|
|
@ -42,6 +43,7 @@ export const LOG_FILTER_LABELS: Record<string, string> = {
|
|||
[LOG_FILTER_IDS.ERROR_MESSAGE]: "Error Message",
|
||||
[LOG_FILTER_IDS.KEY_HASH]: "Key Hash",
|
||||
[LOG_FILTER_IDS.SESSION_ID]: "Session ID",
|
||||
[LOG_FILTER_IDS.CACHE_HIT]: "Cache Hit",
|
||||
[LOG_FILTER_IDS.MODEL_ID]: "Model",
|
||||
[LOG_FILTER_IDS.PUBLIC_MODEL_OR_SEARCH_TOOL]: "Public model / search tool",
|
||||
};
|
||||
|
|
@ -167,6 +169,7 @@ export function useLogFilterLogic({
|
|||
user_id: userIdFilter,
|
||||
end_user: getFilterValue(columnFilters, LOG_FILTER_IDS.END_USER),
|
||||
status_filter: getFilterValue(columnFilters, LOG_FILTER_IDS.STATUS),
|
||||
cache_hit: getFilterValue(columnFilters, LOG_FILTER_IDS.CACHE_HIT),
|
||||
model_id: getFilterValue(columnFilters, LOG_FILTER_IDS.MODEL_ID),
|
||||
model: getFilterValue(columnFilters, LOG_FILTER_IDS.PUBLIC_MODEL_OR_SEARCH_TOOL),
|
||||
key_alias: getFilterValue(columnFilters, LOG_FILTER_IDS.KEY_ALIAS),
|
||||
|
|
|
|||
4
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
4
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
|
|
@ -53273,6 +53273,8 @@ export interface operations {
|
|||
page_size?: number;
|
||||
/** @description Filter logs by status (e.g., success, failure) */
|
||||
status_filter?: string | null;
|
||||
/** @description Filter logs by response cache result: 'true' (served from cache) or 'false' (cache miss) */
|
||||
cache_hit?: string | null;
|
||||
/** @description Filter logs by model */
|
||||
model?: string | null;
|
||||
/** @description Filter logs by model ID (litellm model deployment id) */
|
||||
|
|
@ -53381,6 +53383,8 @@ export interface operations {
|
|||
page_size?: number;
|
||||
/** @description Filter logs by status (e.g., success, failure) */
|
||||
status_filter?: string | null;
|
||||
/** @description Filter logs by response cache result: 'true' (served from cache) or 'false' (cache miss) */
|
||||
cache_hit?: string | null;
|
||||
/** @description Filter logs by model */
|
||||
model?: string | null;
|
||||
/** @description Filter logs by model ID (litellm model deployment id) */
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue