From 26f6ff24d85551dc9edd02b19c341528172496e2 Mon Sep 17 00:00:00 2001
From: Tin Chi Lo
Date: Fri, 24 Jul 2026 17:18:59 -0700
Subject: [PATCH 1/4] fix(proxy): cap tool spend window at 30 days and bound
every SpendLogs read
GET /v1/tool/spend aggregated LiteLLM_SpendLogToolIndex joined to
LiteLLM_SpendLogs with a start_time-only predicate the composite
(tool_name, start_time) index cannot serve, and the dedup total query
left the outer SpendLogs scan unwindowed, so every dashboard load
walked both per-request tables end to end.
- clamp the window to the most recent 30 days ending at end_date; the
response start_date reflects the effective window and the dashboard
notes the clamp
- index SpendLogToolIndex on start_time (all schema copies + migration)
- window the SpendLogs side of both queries (1s margin: the two writers
can disagree by ~1ms on the same request)
- expire SpendLogToolIndex rows on the spend-log retention cutoff via a
parametrized batch-delete engine shared with the SpendLogs cleanup
Co-Authored-By: Claude Fable 5
---
.../migration.sql | 2 +
.../litellm_proxy_extras/schema.prisma | 1 +
litellm/constants.py | 1 +
litellm/proxy/_lazy_openapi_snapshot.json | 2 +-
.../db_transaction_queue/spend_log_cleanup.py | 73 ++++++++++++++-----
.../tool_management_endpoints.py | 19 ++++-
litellm/proxy/schema.prisma | 1 +
schema.prisma | 1 +
.../test_tool_management_endpoints.py | 37 ++++++++++
.../proxy/test_spend_log_cleanup.py | 45 ++++++++++--
.../_components/UsageTab.test.tsx | 27 +++++++
.../_components/UsageTab.tsx | 6 ++
ui/litellm-dashboard/src/lib/http/schema.d.ts | 4 +
13 files changed, 191 insertions(+), 28 deletions(-)
create mode 100644 litellm-proxy-extras/litellm_proxy_extras/migrations/20260724000000_add_spend_log_tool_index_start_time_idx/migration.sql
diff --git a/litellm-proxy-extras/litellm_proxy_extras/migrations/20260724000000_add_spend_log_tool_index_start_time_idx/migration.sql b/litellm-proxy-extras/litellm_proxy_extras/migrations/20260724000000_add_spend_log_tool_index_start_time_idx/migration.sql
new file mode 100644
index 00000000000..548c3bd5683
--- /dev/null
+++ b/litellm-proxy-extras/litellm_proxy_extras/migrations/20260724000000_add_spend_log_tool_index_start_time_idx/migration.sql
@@ -0,0 +1,2 @@
+-- CreateIndex
+CREATE INDEX IF NOT EXISTS "LiteLLM_SpendLogToolIndex_start_time_idx" ON "LiteLLM_SpendLogToolIndex"("start_time");
diff --git a/litellm-proxy-extras/litellm_proxy_extras/schema.prisma b/litellm-proxy-extras/litellm_proxy_extras/schema.prisma
index 23a9c086c73..6713b212314 100644
--- a/litellm-proxy-extras/litellm_proxy_extras/schema.prisma
+++ b/litellm-proxy-extras/litellm_proxy_extras/schema.prisma
@@ -1094,6 +1094,7 @@ model LiteLLM_SpendLogToolIndex {
@@id([request_id, tool_name])
@@index([tool_name, start_time])
+ @@index([start_time])
}
// Prompt table for storing prompt configurations
diff --git a/litellm/constants.py b/litellm/constants.py
index b9b9c0ba604..62d351cffe9 100644
--- a/litellm/constants.py
+++ b/litellm/constants.py
@@ -1455,6 +1455,7 @@ SPEND_LOG_CLEANUP_MAX_CONSECUTIVE_BATCH_FAILURES = int(os.getenv("SPEND_LOG_CLEA
SPEND_LOG_CLEANUP_BATCH_FAILURE_BACKOFF_SECONDS = float(
os.getenv("SPEND_LOG_CLEANUP_BATCH_FAILURE_BACKOFF_SECONDS", 0.5)
)
+TOOL_SPEND_MAX_WINDOW_DAYS = 30
SPEND_LOG_PARTITION_INTERVAL = os.getenv("SPEND_LOG_PARTITION_INTERVAL", "day")
SPEND_LOG_PARTITION_PRECREATE_AHEAD = int(os.getenv("SPEND_LOG_PARTITION_PRECREATE_AHEAD", 7))
SPEND_LOG_QUEUE_SIZE_THRESHOLD = int(os.getenv("SPEND_LOG_QUEUE_SIZE_THRESHOLD", 100))
diff --git a/litellm/proxy/_lazy_openapi_snapshot.json b/litellm/proxy/_lazy_openapi_snapshot.json
index 7e8d08e7cad..6be003927d6 100644
--- a/litellm/proxy/_lazy_openapi_snapshot.json
+++ b/litellm/proxy/_lazy_openapi_snapshot.json
@@ -27417,7 +27417,7 @@
},
"/v1/tool/spend": {
"get": {
- "description": "Spend attributed to each tool over a date range, for the Cost Optimization dashboard.\n\nJoins ``LiteLLM_SpendLogToolIndex`` (which tool names ran on which request) to\n``LiteLLM_SpendLogs`` (what the request cost). A request that used multiple tools\ncounts its full spend toward each of those tools, so per-tool numbers are\nattributions. ``total_spend`` is the deduplicated spend of every request that\ncalled at least one tool in the window, so it never double counts.",
+ "description": "Spend attributed to each tool over a date range, for the Cost Optimization dashboard.\n\nJoins ``LiteLLM_SpendLogToolIndex`` (which tool names ran on which request) to\n``LiteLLM_SpendLogs`` (what the request cost). A request that used multiple tools\ncounts its full spend toward each of those tools, so per-tool numbers are\nattributions. ``total_spend`` is the deduplicated spend of every request that\ncalled at least one tool in the window, so it never double counts.\n\nThe window is capped at the most recent 30 days ending at ``end_date``: a wider\nrequested range is clamped, and the response's ``start_date`` reflects the\neffective window actually served.",
"operationId": "get_tool_spend_v1_tool_spend_get",
"parameters": [
{
diff --git a/litellm/proxy/db/db_transaction_queue/spend_log_cleanup.py b/litellm/proxy/db/db_transaction_queue/spend_log_cleanup.py
index 93f45fd1198..9bf3da0066d 100644
--- a/litellm/proxy/db/db_transaction_queue/spend_log_cleanup.py
+++ b/litellm/proxy/db/db_transaction_queue/spend_log_cleanup.py
@@ -73,30 +73,41 @@ class SpendLogCleanup:
)
return False
- async def _delete_old_logs(self, prisma_client: PrismaClient, cutoff_date: datetime) -> int:
+ async def _delete_old_rows_batched(
+ self,
+ prisma_client: PrismaClient,
+ cutoff_date: datetime,
+ table_name: str,
+ key_columns: tuple[str, ...],
+ time_column: str,
+ ) -> int:
"""
- Helper method to delete old logs in batches.
- Returns the total number of logs deleted.
+ Helper method to delete a table's rows older than the cutoff in batches.
+ Returns the total number of rows deleted.
"""
+ key_list = ", ".join(f'"{col}"' for col in key_columns)
+ delete_sql = f"""
+ DELETE FROM "{table_name}"
+ WHERE ({key_list}) IN (
+ SELECT {key_list} FROM "{table_name}"
+ WHERE "{time_column}" < $1::timestamptz
+ LIMIT $2
+ )
+ """
total_deleted = 0
run_count = 0
consecutive_failures = 0
while True:
if run_count > SPEND_LOG_RUN_LOOPS:
- verbose_proxy_logger.info("Max logs deleted - 1,00,000, rest of the logs will be deleted in next run")
+ verbose_proxy_logger.info(
+ "Max batches reached for %s cleanup, remaining rows will be deleted in next run", table_name
+ )
break
- # Step 1: Find logs and delete them in one go without fetching to application
+ # Step 1: Find rows and delete them in one go without fetching to application
# Delete in batches, limited by self.batch_size
try:
deleted_result = await prisma_client.db.execute_raw(
- """
- DELETE FROM "LiteLLM_SpendLogs"
- WHERE ("request_id", "startTime") IN (
- SELECT "request_id", "startTime" FROM "LiteLLM_SpendLogs"
- WHERE "startTime" < $1::timestamptz
- LIMIT $2
- )
- """,
+ delete_sql,
cutoff_date,
self.batch_size,
)
@@ -105,9 +116,10 @@ class SpendLogCleanup:
# the whole run — subsequent batches may still succeed.
consecutive_failures += 1
verbose_proxy_logger.exception(
- "Spend log cleanup batch failed "
+ "%s cleanup batch failed "
"(run_count=%d, consecutive_failures=%d, batch_size=%d, "
"cutoff=%s, total_deleted_so_far=%d): %s: %s",
+ table_name,
run_count,
consecutive_failures,
self.batch_size,
@@ -118,8 +130,8 @@ class SpendLogCleanup:
)
if consecutive_failures >= SPEND_LOG_CLEANUP_MAX_CONSECUTIVE_BATCH_FAILURES:
verbose_proxy_logger.error(
- "Aborting spend log cleanup after %d consecutive batch "
- "failures; total deleted before abort: %d",
+ "Aborting %s cleanup after %d consecutive batch failures; total deleted before abort: %d",
+ table_name,
consecutive_failures,
total_deleted,
)
@@ -134,15 +146,15 @@ class SpendLogCleanup:
deleted_count = deleted_result
else:
verbose_proxy_logger.error(
- f"Unexpected execute_raw return type for spend log cleanup: {type(deleted_result)}; "
+ f"Unexpected execute_raw return type for {table_name} cleanup: {type(deleted_result)}; "
"aborting cleanup to avoid infinite loop"
)
break
- verbose_proxy_logger.info(f"Deleted {deleted_count} logs in this batch")
+ verbose_proxy_logger.info(f"Deleted {deleted_count} {table_name} rows in this batch")
if deleted_count == 0:
- verbose_proxy_logger.info(f"No more logs to delete. Total deleted: {total_deleted}")
+ verbose_proxy_logger.info(f"No more {table_name} rows to delete. Total deleted: {total_deleted}")
break
total_deleted += deleted_count
@@ -153,6 +165,26 @@ class SpendLogCleanup:
return total_deleted
+ async def _delete_old_logs(self, prisma_client: PrismaClient, cutoff_date: datetime) -> int:
+ return await self._delete_old_rows_batched(
+ prisma_client,
+ cutoff_date,
+ table_name="LiteLLM_SpendLogs",
+ key_columns=("request_id", "startTime"),
+ time_column="startTime",
+ )
+
+ async def _delete_old_tool_index_rows(self, prisma_client: PrismaClient, cutoff_date: datetime) -> int:
+ # SpendLogToolIndex rows are derived from spend logs, so they expire on the
+ # same cutoff; rows older than retention point at already-deleted logs.
+ return await self._delete_old_rows_batched(
+ prisma_client,
+ cutoff_date,
+ table_name="LiteLLM_SpendLogToolIndex",
+ key_columns=("request_id", "tool_name"),
+ time_column="start_time",
+ )
+
async def cleanup_old_spend_logs(self, prisma_client: PrismaClient) -> None:
"""
Main cleanup function. Deletes old spend logs in batches.
@@ -209,6 +241,9 @@ class SpendLogCleanup:
total_deleted = await self._delete_old_logs(prisma_client, cutoff_date)
verbose_proxy_logger.info(f"Deleted {total_deleted} logs")
+ index_deleted = await self._delete_old_tool_index_rows(prisma_client, cutoff_date)
+ verbose_proxy_logger.info(f"Deleted {index_deleted} expired tool index rows")
+
except Exception as e:
# .exception() captures the traceback; str(e) alone on a Prisma/DB
# timeout is often empty and gives operators no signal to diagnose.
diff --git a/litellm/proxy/management_endpoints/tool_management_endpoints.py b/litellm/proxy/management_endpoints/tool_management_endpoints.py
index ca606e07cee..6d9748332bc 100644
--- a/litellm/proxy/management_endpoints/tool_management_endpoints.py
+++ b/litellm/proxy/management_endpoints/tool_management_endpoints.py
@@ -21,6 +21,7 @@ if TYPE_CHECKING:
from litellm.proxy.utils import PrismaClient
from litellm._logging import verbose_proxy_logger
+from litellm.constants import TOOL_SPEND_MAX_WINDOW_DAYS
from litellm.proxy._types import CommonProxyErrors, LitellmUserRoles, UserAPIKeyAuth
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
from litellm.repositories.object_permission_repository import ObjectPermissionRepository
@@ -209,6 +210,10 @@ async def get_tool_spend(
counts its full spend toward each of those tools, so per-tool numbers are
attributions. ``total_spend`` is the deduplicated spend of every request that
called at least one tool in the window, so it never double counts.
+
+ The window is capped at the most recent 30 days ending at ``end_date``: a wider
+ requested range is clamped, and the response's ``start_date`` reflects the
+ effective window actually served.
"""
from litellm.proxy.proxy_server import prisma_client
@@ -226,9 +231,15 @@ async def get_tool_spend(
now = datetime.now(timezone.utc)
end_day = _parse_day_start(end_date)
- start_dt = _parse_day_start(start_date) or ((end_day or now) - timedelta(days=30))
+ window_floor = (end_day or now) - timedelta(days=TOOL_SPEND_MAX_WINDOW_DAYS)
+ start_dt = _parse_day_start(start_date) or window_floor
+ if start_dt < window_floor:
+ start_dt = window_floor
end_exclusive = (end_day + timedelta(days=1)) if end_day else now
+ # ti.start_time defines the window in both queries; the sl."startTime" bounds
+ # exist only so the planner can use the SpendLogs startTime index, and carry a
+ # 1s margin because the two writers can disagree by ~1ms on the same request.
rows = await prisma_client.db.query_raw(
"""
SELECT to_char(ti.start_time, 'YYYY-MM-DD') AS date,
@@ -240,6 +251,8 @@ async def get_tool_spend(
JOIN "LiteLLM_SpendLogs" sl ON sl.request_id = ti.request_id
WHERE ti.start_time >= ($1::timestamptz AT TIME ZONE 'UTC')
AND ti.start_time < ($2::timestamptz AT TIME ZONE 'UTC')
+ AND sl."startTime" >= ($1::timestamptz AT TIME ZONE 'UTC') - interval '1 second'
+ AND sl."startTime" < ($2::timestamptz AT TIME ZONE 'UTC') + interval '1 second'
GROUP BY date, ti.tool_name
ORDER BY date ASC, spend DESC
""",
@@ -250,7 +263,9 @@ async def get_tool_spend(
"""
SELECT COALESCE(SUM(sl.spend), 0)::double precision AS total_spend
FROM "LiteLLM_SpendLogs" sl
- WHERE EXISTS (
+ WHERE sl."startTime" >= ($1::timestamptz AT TIME ZONE 'UTC') - interval '1 second'
+ AND sl."startTime" < ($2::timestamptz AT TIME ZONE 'UTC') + interval '1 second'
+ AND EXISTS (
SELECT 1
FROM "LiteLLM_SpendLogToolIndex" ti
WHERE ti.request_id = sl.request_id
diff --git a/litellm/proxy/schema.prisma b/litellm/proxy/schema.prisma
index 23a9c086c73..6713b212314 100644
--- a/litellm/proxy/schema.prisma
+++ b/litellm/proxy/schema.prisma
@@ -1094,6 +1094,7 @@ model LiteLLM_SpendLogToolIndex {
@@id([request_id, tool_name])
@@index([tool_name, start_time])
+ @@index([start_time])
}
// Prompt table for storing prompt configurations
diff --git a/schema.prisma b/schema.prisma
index 23a9c086c73..6713b212314 100644
--- a/schema.prisma
+++ b/schema.prisma
@@ -1094,6 +1094,7 @@ model LiteLLM_SpendLogToolIndex {
@@id([request_id, tool_name])
@@index([tool_name, start_time])
+ @@index([start_time])
}
// Prompt table for storing prompt configurations
diff --git a/tests/test_litellm/proxy/management_endpoints/test_tool_management_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_tool_management_endpoints.py
index 351f125052d..f420eed6a33 100644
--- a/tests/test_litellm/proxy/management_endpoints/test_tool_management_endpoints.py
+++ b/tests/test_litellm/proxy/management_endpoints/test_tool_management_endpoints.py
@@ -203,6 +203,43 @@ class TestToolManagementEndpoints:
assert tuple(call.args[1:]) == expected_binds
assert resp.json()["end_date"] == "2026-07-02"
+ def test_tool_spend_window_clamped_to_most_recent_30_days(self):
+ prisma = MagicMock()
+ prisma.db.query_raw = AsyncMock(return_value=[])
+ with patch("litellm.proxy.proxy_server.prisma_client", prisma):
+ resp = self.client.get("/v1/tool/spend?start_date=2026-01-01&end_date=2026-07-01")
+ assert resp.status_code == 200
+ expected_binds = (
+ datetime(2026, 6, 1, tzinfo=timezone.utc).isoformat(),
+ datetime(2026, 7, 2, tzinfo=timezone.utc).isoformat(),
+ )
+ assert prisma.db.query_raw.await_count == 2
+ for call in prisma.db.query_raw.await_args_list:
+ assert tuple(call.args[1:]) == expected_binds
+ assert resp.json()["start_date"] == "2026-06-01"
+ assert resp.json()["end_date"] == "2026-07-01"
+
+ def test_tool_spend_range_within_cap_is_not_clamped(self):
+ prisma = MagicMock()
+ prisma.db.query_raw = AsyncMock(return_value=[])
+ with patch("litellm.proxy.proxy_server.prisma_client", prisma):
+ resp = self.client.get("/v1/tool/spend?start_date=2026-06-25&end_date=2026-07-01")
+ assert resp.status_code == 200
+ for call in prisma.db.query_raw.await_args_list:
+ assert call.args[1] == datetime(2026, 6, 25, tzinfo=timezone.utc).isoformat()
+ assert resp.json()["start_date"] == "2026-06-25"
+
+ def test_tool_spend_total_query_bounds_outer_spendlogs_scan(self):
+ prisma = MagicMock()
+ prisma.db.query_raw = AsyncMock(return_value=[])
+ with patch("litellm.proxy.proxy_server.prisma_client", prisma):
+ resp = self.client.get("/v1/tool/spend?start_date=2026-07-01&end_date=2026-07-02")
+ assert resp.status_code == 200
+ for call in prisma.db.query_raw.await_args_list:
+ sql = call.args[0]
+ assert 'sl."startTime" >=' in sql
+ assert 'sl."startTime" <' in sql
+
@pytest.mark.parametrize(
"query",
[
diff --git a/tests/test_litellm/proxy/test_spend_log_cleanup.py b/tests/test_litellm/proxy/test_spend_log_cleanup.py
index a309dd64011..23621292b6e 100644
--- a/tests/test_litellm/proxy/test_spend_log_cleanup.py
+++ b/tests/test_litellm/proxy/test_spend_log_cleanup.py
@@ -157,8 +157,9 @@ async def test_cleanup_old_spend_logs_batch_deletion():
mock_prisma_client = MagicMock()
mock_db = MagicMock()
- # Mock execute_raw to return deleted counts
- mock_db.execute_raw = AsyncMock(side_effect=[1000, 500, 0])
+ # Mock execute_raw to return deleted counts (3 spend-log batches, then the
+ # tool-index cleanup's first batch returning 0)
+ mock_db.execute_raw = AsyncMock(side_effect=[1000, 500, 0, 0])
# Wire up mocks
mock_prisma_client.db = mock_db
@@ -178,7 +179,7 @@ async def test_cleanup_old_spend_logs_batch_deletion():
await cleaner.cleanup_old_spend_logs(mock_prisma_client)
# Validate batching and deletion via raw SQL
- assert mock_db.execute_raw.call_count == 3
+ assert mock_db.execute_raw.call_count == 4
# Check the first call argument
call_args_sql = mock_db.execute_raw.call_args_list[0][0][0]
@@ -188,6 +189,10 @@ async def test_cleanup_old_spend_logs_batch_deletion():
# reusing x-litellm-call-id take out a fresh row alongside the expired one
assert 'WHERE ("request_id", "startTime") IN' in call_args_sql
+ # After spend logs, the derived tool index rows expire on the same cutoff
+ tool_index_sql = mock_db.execute_raw.call_args_list[3][0][0]
+ assert 'DELETE FROM "LiteLLM_SpendLogToolIndex"' in tool_index_sql
+
@pytest.mark.asyncio
async def test_cleanup_old_spend_logs_retention_period_cutoff():
@@ -258,6 +263,10 @@ async def test_cleanup_drops_partitions_when_enabled_and_partitioned():
partition_manager.drop_partitions_older_than.assert_awaited_once()
delete_sql = mock_prisma_client.db.execute_raw.call_args_list[0][0][0]
assert 'DELETE FROM "LiteLLM_SpendLogs"' in delete_sql
+ # Partition drops only reclaim spend logs; the tool index must still be
+ # cleaned row-wise on the same run
+ all_sql = [c[0][0] for c in mock_prisma_client.db.execute_raw.call_args_list]
+ assert any('DELETE FROM "LiteLLM_SpendLogToolIndex"' in s for s in all_sql)
@pytest.mark.asyncio
@@ -270,7 +279,7 @@ async def test_cleanup_uses_delete_when_partitioning_not_enabled():
from unittest.mock import AsyncMock, MagicMock
mock_prisma_client = MagicMock()
- mock_prisma_client.db.execute_raw = AsyncMock(side_effect=[10, 0])
+ mock_prisma_client.db.execute_raw = AsyncMock(side_effect=[10, 0, 0])
partition_manager = MagicMock()
partition_manager.is_partitioned = AsyncMock(return_value=True)
@@ -301,7 +310,7 @@ async def test_cleanup_uses_delete_when_not_partitioned():
from unittest.mock import AsyncMock, MagicMock
mock_prisma_client = MagicMock()
- mock_prisma_client.db.execute_raw = AsyncMock(side_effect=[10, 0])
+ mock_prisma_client.db.execute_raw = AsyncMock(side_effect=[10, 0, 0])
partition_manager = MagicMock()
partition_manager.is_partitioned = AsyncMock(return_value=False)
@@ -320,7 +329,7 @@ async def test_cleanup_uses_delete_when_not_partitioned():
await cleaner.cleanup_old_spend_logs(mock_prisma_client)
partition_manager.drop_partitions_older_than.assert_not_awaited()
- assert mock_prisma_client.db.execute_raw.await_count == 2
+ assert mock_prisma_client.db.execute_raw.await_count == 3
delete_sql = mock_prisma_client.db.execute_raw.call_args_list[0][0][0]
assert 'DELETE FROM "LiteLLM_SpendLogs"' in delete_sql
@@ -437,6 +446,30 @@ async def test_delete_old_logs_continues_on_valid_int_return():
assert total_deleted == 800
+@pytest.mark.asyncio
+async def test_delete_old_tool_index_rows_deletes_on_composite_key():
+ """Tool index rows are derived from spend logs and expire on the same cutoff;
+ the delete must match on the table's composite primary key."""
+ mock_prisma_client = MagicMock()
+ mock_db = MagicMock()
+ mock_db.execute_raw = AsyncMock(side_effect=[5, 0])
+ mock_prisma_client.db = mock_db
+
+ cleaner = SpendLogCleanup(
+ general_settings={"maximum_spend_logs_retention_period": "7d"}
+ )
+
+ cutoff_date = datetime.now(timezone.utc) - timedelta(days=7)
+ total_deleted = await cleaner._delete_old_tool_index_rows(mock_prisma_client, cutoff_date)
+
+ assert total_deleted == 5
+ delete_sql = mock_db.execute_raw.call_args_list[0][0][0]
+ assert 'DELETE FROM "LiteLLM_SpendLogToolIndex"' in delete_sql
+ assert 'WHERE ("request_id", "tool_name") IN' in delete_sql
+ assert '"start_time" <' in delete_sql
+ assert mock_db.execute_raw.call_args_list[0][0][1] == cutoff_date
+
+
@pytest.mark.asyncio
async def test_delete_old_logs_continues_after_single_batch_failure(monkeypatch):
"""A single batch failure (e.g. DB timeout) must not abort the whole run —
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/UsageTab.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/UsageTab.test.tsx
index 0e2f16c5d93..27c79c8a9ce 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/UsageTab.test.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/UsageTab.test.tsx
@@ -137,4 +137,31 @@ describe("UsageTab", () => {
const series = JSON.parse(bars[0].getAttribute("data-series") ?? "[]");
expect(series[0]).toMatchObject({ tool_name: "search", spend: 4.0 });
});
+
+ it("notes the 30-day cap when the server clamps the tool spend window", async () => {
+ const toolSpend = {
+ by_tool: [{ tool_name: "search", spend: 4.0, call_count: 3, total_tokens: 150 }],
+ daily: [{ date: "2026-07-12", tool_name: "search", spend: 4.0, call_count: 3 }],
+ total_spend: 4.0,
+ start_date: "2026-07-05",
+ end_date: "2026-07-14",
+ };
+ const { findByText } = renderWith([day("2026-07-12", {})], toolSpend);
+
+ expect(await findByText(/most recent 30 days of the selected range/)).toBeInTheDocument();
+ });
+
+ it("shows no cap note when the served window matches the request", async () => {
+ const toolSpend = {
+ by_tool: [{ tool_name: "search", spend: 4.0, call_count: 3, total_tokens: 150 }],
+ daily: [{ date: "2026-07-12", tool_name: "search", spend: 4.0, call_count: 3 }],
+ total_spend: 4.0,
+ start_date: "2026-07-01",
+ end_date: "2026-07-14",
+ };
+ const { findAllByTestId, queryByText } = renderWith([day("2026-07-12", {})], toolSpend);
+
+ await findAllByTestId("bar-chart");
+ expect(queryByText(/most recent 30 days of the selected range/)).not.toBeInTheDocument();
+ });
});
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/UsageTab.tsx b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/UsageTab.tsx
index 9216dbfaad2..a069c80a201 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/UsageTab.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/UsageTab.tsx
@@ -108,6 +108,7 @@ const UsageTab: React.FC = ({ accessToken, activity }) => {
const toolSpend = toolSpendState?.key === rangeKey ? toolSpendState.data : null;
const toolSpendLoading = toolSpendEnabled && toolSpend === null;
+ const toolSpendWindowClamped = !!toolSpend?.start_date && !!startTime && toolSpend.start_date > isoDay(startTime);
const compressionTotal = useMemo(() => results.reduce((sum, d) => sum + compressionOf(d.metrics), 0), [results]);
const cachingTotal = useMemo(() => results.reduce((sum, d) => sum + cachingOf(d.metrics), 0), [results]);
@@ -211,6 +212,11 @@ const UsageTab: React.FC = ({ accessToken, activity }) => {
Spend on requests that called each tool (MCP and client-side tools). A request that used multiple tools
counts its full spend toward each, so this attributes rather than partitions spend.
+ {toolSpendWindowClamped && (
+
+ Tool spend is limited to the most recent 30 days of the selected range (since {toolSpend?.start_date}).
+
+ )}
{topTools.length === 0 ? (
diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts
index aafab811b83..e420da55e77 100644
--- a/ui/litellm-dashboard/src/lib/http/schema.d.ts
+++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts
@@ -17976,6 +17976,10 @@ export interface paths {
* counts its full spend toward each of those tools, so per-tool numbers are
* attributions. ``total_spend`` is the deduplicated spend of every request that
* called at least one tool in the window, so it never double counts.
+ *
+ * The window is capped at the most recent 30 days ending at ``end_date``: a wider
+ * requested range is clamped, and the response's ``start_date`` reflects the
+ * effective window actually served.
*/
get: operations["get_tool_spend_v1_tool_spend_get"];
put?: never;
From 30b7fd16f0f75251e72f3e87a54d97d3550893f7 Mon Sep 17 00:00:00 2001
From: Tin Chi Lo
Date: Fri, 24 Jul 2026 17:31:00 -0700
Subject: [PATCH 2/4] fix(proxy): label the tool spend clamp accurately (start
capped at 30 days before end)
The clamp floor is end_date minus 30 days, serving up to 31 calendar
dates inclusive: deliberately the same width as the endpoint's default
window, so the dashboard's own default range never triggers the clamp
note. The docstring, card note, and test name now state that invariant
instead of the misleading 'most recent 30 days'.
Co-Authored-By: Claude Fable 5
---
litellm/proxy/_lazy_openapi_snapshot.json | 2 +-
.../management_endpoints/tool_management_endpoints.py | 7 ++++---
.../management_endpoints/test_tool_management_endpoints.py | 5 ++++-
.../cost-optimization/_components/UsageTab.test.tsx | 4 ++--
.../(dashboard)/cost-optimization/_components/UsageTab.tsx | 3 ++-
ui/litellm-dashboard/src/lib/http/schema.d.ts | 7 ++++---
6 files changed, 17 insertions(+), 11 deletions(-)
diff --git a/litellm/proxy/_lazy_openapi_snapshot.json b/litellm/proxy/_lazy_openapi_snapshot.json
index 6be003927d6..972c831073f 100644
--- a/litellm/proxy/_lazy_openapi_snapshot.json
+++ b/litellm/proxy/_lazy_openapi_snapshot.json
@@ -27417,7 +27417,7 @@
},
"/v1/tool/spend": {
"get": {
- "description": "Spend attributed to each tool over a date range, for the Cost Optimization dashboard.\n\nJoins ``LiteLLM_SpendLogToolIndex`` (which tool names ran on which request) to\n``LiteLLM_SpendLogs`` (what the request cost). A request that used multiple tools\ncounts its full spend toward each of those tools, so per-tool numbers are\nattributions. ``total_spend`` is the deduplicated spend of every request that\ncalled at least one tool in the window, so it never double counts.\n\nThe window is capped at the most recent 30 days ending at ``end_date``: a wider\nrequested range is clamped, and the response's ``start_date`` reflects the\neffective window actually served.",
+ "description": "Spend attributed to each tool over a date range, for the Cost Optimization dashboard.\n\nJoins ``LiteLLM_SpendLogToolIndex`` (which tool names ran on which request) to\n``LiteLLM_SpendLogs`` (what the request cost). A request that used multiple tools\ncounts its full spend toward each of those tools, so per-tool numbers are\nattributions. ``total_spend`` is the deduplicated spend of every request that\ncalled at least one tool in the window, so it never double counts.\n\n``start_date`` is clamped to at most 30 days before ``end_date`` (serving up to\n31 calendar dates inclusive, the same width as the endpoint's default window):\na wider requested range is clamped, and the response's ``start_date`` reflects\nthe effective window actually served.",
"operationId": "get_tool_spend_v1_tool_spend_get",
"parameters": [
{
diff --git a/litellm/proxy/management_endpoints/tool_management_endpoints.py b/litellm/proxy/management_endpoints/tool_management_endpoints.py
index 6d9748332bc..3f528c8c489 100644
--- a/litellm/proxy/management_endpoints/tool_management_endpoints.py
+++ b/litellm/proxy/management_endpoints/tool_management_endpoints.py
@@ -211,9 +211,10 @@ async def get_tool_spend(
attributions. ``total_spend`` is the deduplicated spend of every request that
called at least one tool in the window, so it never double counts.
- The window is capped at the most recent 30 days ending at ``end_date``: a wider
- requested range is clamped, and the response's ``start_date`` reflects the
- effective window actually served.
+ ``start_date`` is clamped to at most 30 days before ``end_date`` (serving up to
+ 31 calendar dates inclusive, the same width as the endpoint's default window):
+ a wider requested range is clamped, and the response's ``start_date`` reflects
+ the effective window actually served.
"""
from litellm.proxy.proxy_server import prisma_client
diff --git a/tests/test_litellm/proxy/management_endpoints/test_tool_management_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_tool_management_endpoints.py
index f420eed6a33..150499dfbf1 100644
--- a/tests/test_litellm/proxy/management_endpoints/test_tool_management_endpoints.py
+++ b/tests/test_litellm/proxy/management_endpoints/test_tool_management_endpoints.py
@@ -203,7 +203,10 @@ class TestToolManagementEndpoints:
assert tuple(call.args[1:]) == expected_binds
assert resp.json()["end_date"] == "2026-07-02"
- def test_tool_spend_window_clamped_to_most_recent_30_days(self):
+ def test_tool_spend_start_clamped_to_30_days_before_end(self):
+ # Clamped floor is end_date minus 30 days, serving up to 31 calendar dates
+ # inclusive: deliberately the same width as the endpoint's default window,
+ # so the dashboard's default range never triggers the clamp.
prisma = MagicMock()
prisma.db.query_raw = AsyncMock(return_value=[])
with patch("litellm.proxy.proxy_server.prisma_client", prisma):
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/UsageTab.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/UsageTab.test.tsx
index 27c79c8a9ce..ab1cd146adc 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/UsageTab.test.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/UsageTab.test.tsx
@@ -148,7 +148,7 @@ describe("UsageTab", () => {
};
const { findByText } = renderWith([day("2026-07-12", {})], toolSpend);
- expect(await findByText(/most recent 30 days of the selected range/)).toBeInTheDocument();
+ expect(await findByText(/capped at 30 days before the end of the selected range/)).toBeInTheDocument();
});
it("shows no cap note when the served window matches the request", async () => {
@@ -162,6 +162,6 @@ describe("UsageTab", () => {
const { findAllByTestId, queryByText } = renderWith([day("2026-07-12", {})], toolSpend);
await findAllByTestId("bar-chart");
- expect(queryByText(/most recent 30 days of the selected range/)).not.toBeInTheDocument();
+ expect(queryByText(/capped at 30 days before the end of the selected range/)).not.toBeInTheDocument();
});
});
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/UsageTab.tsx b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/UsageTab.tsx
index a069c80a201..6e2da4456d3 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/UsageTab.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/UsageTab.tsx
@@ -214,7 +214,8 @@ const UsageTab: React.FC = ({ accessToken, activity }) => {
{toolSpendWindowClamped && (
- Tool spend is limited to the most recent 30 days of the selected range (since {toolSpend?.start_date}).
+ Tool spend is capped at 30 days before the end of the selected range; showing spend since{" "}
+ {toolSpend?.start_date}.
)}
diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts
index e420da55e77..c1141db7852 100644
--- a/ui/litellm-dashboard/src/lib/http/schema.d.ts
+++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts
@@ -17977,9 +17977,10 @@ export interface paths {
* attributions. ``total_spend`` is the deduplicated spend of every request that
* called at least one tool in the window, so it never double counts.
*
- * The window is capped at the most recent 30 days ending at ``end_date``: a wider
- * requested range is clamped, and the response's ``start_date`` reflects the
- * effective window actually served.
+ * ``start_date`` is clamped to at most 30 days before ``end_date`` (serving up to
+ * 31 calendar dates inclusive, the same width as the endpoint's default window):
+ * a wider requested range is clamped, and the response's ``start_date`` reflects
+ * the effective window actually served.
*/
get: operations["get_tool_spend_v1_tool_spend_get"];
put?: never;
From 5fc4a058c1cabd30d22909ecf4c25b05b757e352 Mon Sep 17 00:00:00 2001
From: Tin Chi Lo
Date: Fri, 24 Jul 2026 17:43:15 -0700
Subject: [PATCH 3/4] test(proxy): cover the cleanup run-loop backstop branch
Co-Authored-By: Claude Fable 5
---
.../proxy/test_spend_log_cleanup.py | 25 +++++++++++++++++++
1 file changed, 25 insertions(+)
diff --git a/tests/test_litellm/proxy/test_spend_log_cleanup.py b/tests/test_litellm/proxy/test_spend_log_cleanup.py
index 23621292b6e..f969b040a0d 100644
--- a/tests/test_litellm/proxy/test_spend_log_cleanup.py
+++ b/tests/test_litellm/proxy/test_spend_log_cleanup.py
@@ -446,6 +446,31 @@ async def test_delete_old_logs_continues_on_valid_int_return():
assert total_deleted == 800
+@pytest.mark.asyncio
+async def test_delete_old_rows_stops_at_max_batches(monkeypatch):
+ """The run-loop backstop must halt a cleanup that keeps finding rows, so a
+ huge backlog is spread across scheduled runs instead of one unbounded loop."""
+ import litellm.proxy.db.db_transaction_queue.spend_log_cleanup as cleanup_module
+
+ monkeypatch.setattr(cleanup_module, "SPEND_LOG_RUN_LOOPS", 2)
+
+ mock_prisma_client = MagicMock()
+ mock_db = MagicMock()
+ mock_db.execute_raw = AsyncMock(return_value=1000)
+ mock_prisma_client.db = mock_db
+
+ cleaner = SpendLogCleanup(
+ general_settings={"maximum_spend_logs_retention_period": "7d"}
+ )
+
+ cutoff_date = datetime.now(timezone.utc) - timedelta(days=7)
+ total_deleted = await cleaner._delete_old_logs(mock_prisma_client, cutoff_date)
+
+ # run_count exceeds the cap only after 3 full batches (0, 1, 2)
+ assert mock_db.execute_raw.call_count == 3
+ assert total_deleted == 3000
+
+
@pytest.mark.asyncio
async def test_delete_old_tool_index_rows_deletes_on_composite_key():
"""Tool index rows are derived from spend logs and expire on the same cutoff;
From 3ca9d551dcd8356eb5c1aa1097c884fc066edcaa Mon Sep 17 00:00:00 2001
From: Tin Chi Lo
Date: Fri, 24 Jul 2026 18:00:16 -0700
Subject: [PATCH 4/4] fix(proxy): anchor the tool spend clamp floor to UTC
midnight
With end_date omitted the floor was end-anchored to now including its
time-of-day, so an explicit start_date exactly 30 days back parsed as
midnight, compared below the floor, and was invisibly clamped to a
mid-day instant: up to a day of spend disappeared while the response
start_date still printed the full calendar date. Anchoring the floor to
today's UTC midnight makes every comparison in the window derivation
date-pure.
Co-Authored-By: Claude Fable 5
---
.../tool_management_endpoints.py | 6 ++++-
.../test_tool_management_endpoints.py | 27 ++++++++++++++++++-
2 files changed, 31 insertions(+), 2 deletions(-)
diff --git a/litellm/proxy/management_endpoints/tool_management_endpoints.py b/litellm/proxy/management_endpoints/tool_management_endpoints.py
index 3f528c8c489..b6a445ef327 100644
--- a/litellm/proxy/management_endpoints/tool_management_endpoints.py
+++ b/litellm/proxy/management_endpoints/tool_management_endpoints.py
@@ -232,7 +232,11 @@ async def get_tool_spend(
now = datetime.now(timezone.utc)
end_day = _parse_day_start(end_date)
- window_floor = (end_day or now) - timedelta(days=TOOL_SPEND_MAX_WINDOW_DAYS)
+ # Anchor the floor to a midnight so the clamp compares dates with dates:
+ # parsed start_dates are midnight-aligned, and a floor carrying now's
+ # time-of-day would invisibly truncate an explicit start_date to mid-day.
+ today = now.replace(hour=0, minute=0, second=0, microsecond=0)
+ window_floor = (end_day or today) - timedelta(days=TOOL_SPEND_MAX_WINDOW_DAYS)
start_dt = _parse_day_start(start_date) or window_floor
if start_dt < window_floor:
start_dt = window_floor
diff --git a/tests/test_litellm/proxy/management_endpoints/test_tool_management_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_tool_management_endpoints.py
index 150499dfbf1..c908250fa64 100644
--- a/tests/test_litellm/proxy/management_endpoints/test_tool_management_endpoints.py
+++ b/tests/test_litellm/proxy/management_endpoints/test_tool_management_endpoints.py
@@ -9,7 +9,7 @@ imports these inside function bodies to avoid circular imports.
import os
import sys
-from datetime import datetime, timezone
+from datetime import datetime, timedelta, timezone
from typing import Optional
from unittest.mock import AsyncMock, MagicMock, patch
@@ -232,6 +232,31 @@ class TestToolManagementEndpoints:
assert call.args[1] == datetime(2026, 6, 25, tzinfo=timezone.utc).isoformat()
assert resp.json()["start_date"] == "2026-06-25"
+ def test_tool_spend_start_honored_when_end_date_omitted(self):
+ # Regression: with end_date omitted the floor anchors to today's UTC
+ # midnight, not now's time-of-day, so an explicit start_date exactly 30
+ # days back is served from midnight rather than truncated to mid-day.
+ prisma = MagicMock()
+ prisma.db.query_raw = AsyncMock(return_value=[])
+ floor_day = datetime.now(timezone.utc).replace(hour=0, minute=0, second=0, microsecond=0) - timedelta(days=30)
+ with patch("litellm.proxy.proxy_server.prisma_client", prisma):
+ resp = self.client.get(f"/v1/tool/spend?start_date={floor_day.strftime('%Y-%m-%d')}")
+ assert resp.status_code == 200
+ for call in prisma.db.query_raw.await_args_list:
+ assert call.args[1] == floor_day.isoformat()
+ assert resp.json()["start_date"] == floor_day.strftime("%Y-%m-%d")
+
+ def test_tool_spend_clamp_without_end_date_lands_on_midnight(self):
+ prisma = MagicMock()
+ prisma.db.query_raw = AsyncMock(return_value=[])
+ floor_day = datetime.now(timezone.utc).replace(hour=0, minute=0, second=0, microsecond=0) - timedelta(days=30)
+ with patch("litellm.proxy.proxy_server.prisma_client", prisma):
+ resp = self.client.get("/v1/tool/spend?start_date=2020-01-01")
+ assert resp.status_code == 200
+ for call in prisma.db.query_raw.await_args_list:
+ assert call.args[1] == floor_day.isoformat()
+ assert resp.json()["start_date"] == floor_day.strftime("%Y-%m-%d")
+
def test_tool_spend_total_query_bounds_outer_spendlogs_scan(self):
prisma = MagicMock()
prisma.db.query_raw = AsyncMock(return_value=[])