Speed up /team/daily/activity by removing redundant GROUP BY and adding date-first indexes

The aggregated SQL query used GROUP BY on nearly all columns matching the
unique constraint, making it a no-op that forced PostgreSQL to pay full
hash/sort cost. Replace with a plain SELECT since the unique constraint
already ensures row uniqueness; Python _aggregate_spend_records handles
the final rollup.

Add date-first composite indexes (date, entity_id) to all daily spend
tables to allow efficient B-tree range scans for the common query pattern
of date >= X AND date <= Y AND entity_id NOT IN (...).

Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
yuneng-jiang 2026-03-06 22:35:16 -08:00
parent 39c5adcff1
commit 38be8eaafa
4 changed files with 274 additions and 18 deletions

View file

@ -623,6 +623,7 @@ model LiteLLM_DailyUserSpend {
@@unique([user_id, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint])
@@index([date])
@@index([user_id, date])
@@index([date, user_id])
@@index([api_key])
@@index([model])
@@index([mcp_namespaced_tool_name])
@ -654,6 +655,7 @@ model LiteLLM_DailyOrganizationSpend {
@@unique([organization_id, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint])
@@index([date])
@@index([organization_id, date])
@@index([date, organization_id])
@@index([api_key])
@@index([model])
@@index([mcp_namespaced_tool_name])
@ -684,6 +686,7 @@ model LiteLLM_DailyEndUserSpend {
@@unique([end_user_id, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint])
@@index([date])
@@index([end_user_id, date])
@@index([date, end_user_id])
@@index([api_key])
@@index([model])
@@index([mcp_namespaced_tool_name])
@ -714,6 +717,7 @@ model LiteLLM_DailyAgentSpend {
@@unique([agent_id, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint])
@@index([date])
@@index([agent_id, date])
@@index([date, agent_id])
@@index([api_key])
@@index([model])
@@index([mcp_namespaced_tool_name])
@ -745,6 +749,7 @@ model LiteLLM_DailyTeamSpend {
@@unique([team_id, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint])
@@index([date])
@@index([team_id, date])
@@index([date, team_id])
@@index([api_key])
@@index([model])
@@index([mcp_namespaced_tool_name])
@ -777,6 +782,7 @@ model LiteLLM_DailyTagSpend {
@@unique([tag, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint])
@@index([date])
@@index([tag, date])
@@index([date, tag])
@@index([api_key])
@@index([model])
@@index([mcp_namespaced_tool_name])

View file

@ -479,16 +479,18 @@ def _build_aggregated_sql_query(
timezone_offset_minutes: Optional[int] = None,
include_entity_id: bool = False,
) -> Tuple[str, List[Any]]:
"""Build a parameterized SQL GROUP BY query for aggregated daily activity.
"""Build a parameterized SQL query for aggregated daily activity.
Groups by (date, api_key, model, model_group, custom_llm_provider,
mcp_namespaced_tool_name, endpoint) with SUMs on all metric columns.
Uses a plain SELECT (no GROUP BY) because the table's unique constraint
already ensures near-uniqueness across (entity_id, date, api_key, model,
custom_llm_provider, mcp_namespaced_tool_name, endpoint). Skipping the
GROUP BY avoids the expensive hash/sort aggregation step in PostgreSQL,
which is essentially a no-op on these tables.
When include_entity_id is False (default), the entity_id column is omitted
from GROUP BY to collapse rows across entities.
The Python _aggregate_spend_records function handles the final rollup.
When include_entity_id is True, the entity_id column is included in both
SELECT and GROUP BY, preserving per-entity breakdown in the results.
When include_entity_id is True, the entity_id column is included in SELECT
to preserve per-entity breakdown in the results.
Returns:
Tuple of (sql_query, params_list) ready for prisma_client.db.query_raw().
@ -556,7 +558,6 @@ def _build_aggregated_sql_query(
where_clause = " AND ".join(sql_conditions)
entity_select = f'"{entity_id_field}",' if include_entity_id else ""
entity_group_by = f'"{entity_id_field}",' if include_entity_id else ""
sql_query = f"""
SELECT
@ -568,18 +569,16 @@ def _build_aggregated_sql_query(
custom_llm_provider,
mcp_namespaced_tool_name,
endpoint,
SUM(spend)::float AS spend,
SUM(prompt_tokens)::bigint AS prompt_tokens,
SUM(completion_tokens)::bigint AS completion_tokens,
SUM(cache_read_input_tokens)::bigint AS cache_read_input_tokens,
SUM(cache_creation_input_tokens)::bigint AS cache_creation_input_tokens,
SUM(api_requests)::bigint AS api_requests,
SUM(successful_requests)::bigint AS successful_requests,
SUM(failed_requests)::bigint AS failed_requests
spend::float AS spend,
prompt_tokens::bigint AS prompt_tokens,
completion_tokens::bigint AS completion_tokens,
cache_read_input_tokens::bigint AS cache_read_input_tokens,
cache_creation_input_tokens::bigint AS cache_creation_input_tokens,
api_requests::bigint AS api_requests,
successful_requests::bigint AS successful_requests,
failed_requests::bigint AS failed_requests
FROM "{pg_table}"
WHERE {where_clause}
GROUP BY {entity_group_by} date, api_key, model, model_group, custom_llm_provider,
mcp_namespaced_tool_name, endpoint
ORDER BY date DESC
"""

View file

@ -661,6 +661,7 @@ model LiteLLM_DailyUserSpend {
@@unique([user_id, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint])
@@index([date])
@@index([user_id, date])
@@index([date, user_id])
@@index([api_key])
@@index([model])
@@index([mcp_namespaced_tool_name])
@ -692,6 +693,7 @@ model LiteLLM_DailyOrganizationSpend {
@@unique([organization_id, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint])
@@index([date])
@@index([organization_id, date])
@@index([date, organization_id])
@@index([api_key])
@@index([model])
@@index([mcp_namespaced_tool_name])
@ -722,6 +724,7 @@ model LiteLLM_DailyEndUserSpend {
@@unique([end_user_id, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint])
@@index([date])
@@index([end_user_id, date])
@@index([date, end_user_id])
@@index([api_key])
@@index([model])
@@index([mcp_namespaced_tool_name])
@ -752,6 +755,7 @@ model LiteLLM_DailyAgentSpend {
@@unique([agent_id, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint])
@@index([date])
@@index([agent_id, date])
@@index([date, agent_id])
@@index([api_key])
@@index([model])
@@index([mcp_namespaced_tool_name])
@ -783,6 +787,7 @@ model LiteLLM_DailyTeamSpend {
@@unique([team_id, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint])
@@index([date])
@@index([team_id, date])
@@index([date, team_id])
@@index([api_key])
@@index([model])
@@index([mcp_namespaced_tool_name])
@ -815,6 +820,7 @@ model LiteLLM_DailyTagSpend {
@@unique([tag, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint])
@@index([date])
@@index([tag, date])
@@index([date, tag])
@@index([api_key])
@@index([model])
@@index([mcp_namespaced_tool_name])

View file

@ -9,6 +9,7 @@ sys.path.insert(
) # Adds the parent directory to the system path
from litellm.proxy.management_endpoints.common_daily_activity import (
_build_aggregated_sql_query,
_is_user_agent_tag,
compute_tag_metadata_totals,
get_api_key_metadata,
@ -469,3 +470,247 @@ async def test_aggregated_activity_preserves_metadata_for_deleted_keys():
assert key_data.metadata.key_alias == "toto-test-2"
assert key_data.metadata.team_id == "69cd4b77-b095-4489-8c46-4f2f31d840a2"
assert key_data.metrics.spend == 10.0
class TestBuildAggregatedSqlQuery:
"""Tests for _build_aggregated_sql_query to verify the plain SELECT (no GROUP BY) approach."""
def test_basic_query_no_group_by(self):
"""Verify the query uses plain SELECT without GROUP BY."""
sql, params = _build_aggregated_sql_query(
table_name="litellm_dailyteamspend",
entity_id_field="team_id",
entity_id=None,
start_date="2024-01-01",
end_date="2024-01-31",
model=None,
api_key=None,
)
assert "GROUP BY" not in sql
assert "SUM(" not in sql
assert "ORDER BY date DESC" in sql
assert '"LiteLLM_DailyTeamSpend"' in sql
assert params == ["2024-01-01", "2024-01-31"]
def test_query_selects_columns_directly(self):
"""Verify metric columns are selected with casts, not aggregated."""
sql, _ = _build_aggregated_sql_query(
table_name="litellm_dailyteamspend",
entity_id_field="team_id",
entity_id=None,
start_date="2024-01-01",
end_date="2024-01-31",
model=None,
api_key=None,
)
assert "spend::float AS spend" in sql
assert "prompt_tokens::bigint AS prompt_tokens" in sql
assert "completion_tokens::bigint AS completion_tokens" in sql
assert "api_requests::bigint AS api_requests" in sql
def test_include_entity_id_true(self):
"""When include_entity_id=True, entity column should appear in SELECT."""
sql, params = _build_aggregated_sql_query(
table_name="litellm_dailyteamspend",
entity_id_field="team_id",
entity_id=None,
start_date="2024-01-01",
end_date="2024-01-31",
model=None,
api_key=None,
include_entity_id=True,
)
assert '"team_id",' in sql
assert "GROUP BY" not in sql
def test_include_entity_id_false(self):
"""When include_entity_id=False (default), entity column should not appear."""
sql, _ = _build_aggregated_sql_query(
table_name="litellm_dailyteamspend",
entity_id_field="team_id",
entity_id=None,
start_date="2024-01-01",
end_date="2024-01-31",
model=None,
api_key=None,
include_entity_id=False,
)
assert '"team_id",' not in sql
def test_entity_id_single_value_filter(self):
"""Single entity_id value should produce an = condition."""
sql, params = _build_aggregated_sql_query(
table_name="litellm_dailyteamspend",
entity_id_field="team_id",
entity_id="team-abc",
start_date="2024-01-01",
end_date="2024-01-31",
model=None,
api_key=None,
)
assert '"team_id" = $3' in sql
assert params == ["2024-01-01", "2024-01-31", "team-abc"]
def test_entity_id_list_filter(self):
"""List of entity_ids should produce an IN condition."""
sql, params = _build_aggregated_sql_query(
table_name="litellm_dailyteamspend",
entity_id_field="team_id",
entity_id=["team-a", "team-b"],
start_date="2024-01-01",
end_date="2024-01-31",
model=None,
api_key=None,
)
assert '"team_id" IN ($3, $4)' in sql
assert params == ["2024-01-01", "2024-01-31", "team-a", "team-b"]
def test_exclude_entity_ids(self):
"""exclude_entity_ids should produce a NOT IN condition."""
sql, params = _build_aggregated_sql_query(
table_name="litellm_dailyteamspend",
entity_id_field="team_id",
entity_id=None,
start_date="2024-01-01",
end_date="2024-01-31",
model=None,
api_key=None,
exclude_entity_ids=["team-x", "team-y"],
)
assert '"team_id" NOT IN ($3, $4)' in sql
assert params == ["2024-01-01", "2024-01-31", "team-x", "team-y"]
def test_model_filter(self):
"""model filter should add a model = condition."""
sql, params = _build_aggregated_sql_query(
table_name="litellm_dailyteamspend",
entity_id_field="team_id",
entity_id=None,
start_date="2024-01-01",
end_date="2024-01-31",
model="gpt-4",
api_key=None,
)
assert "model = $3" in sql
assert params == ["2024-01-01", "2024-01-31", "gpt-4"]
def test_api_key_single_filter(self):
"""Single api_key should produce an = condition."""
sql, params = _build_aggregated_sql_query(
table_name="litellm_dailyteamspend",
entity_id_field="team_id",
entity_id=None,
start_date="2024-01-01",
end_date="2024-01-31",
model=None,
api_key="key-123",
)
assert "api_key = $3" in sql
assert params == ["2024-01-01", "2024-01-31", "key-123"]
def test_api_key_list_filter(self):
"""List of api_keys should produce an IN condition."""
sql, params = _build_aggregated_sql_query(
table_name="litellm_dailyteamspend",
entity_id_field="team_id",
entity_id=None,
start_date="2024-01-01",
end_date="2024-01-31",
model=None,
api_key=["key-1", "key-2"],
)
assert "api_key IN ($3, $4)" in sql
assert params == ["2024-01-01", "2024-01-31", "key-1", "key-2"]
def test_all_filters_combined(self):
"""All filters together should produce correct parameter ordering."""
sql, params = _build_aggregated_sql_query(
table_name="litellm_dailyteamspend",
entity_id_field="team_id",
entity_id="team-abc",
start_date="2024-01-01",
end_date="2024-01-31",
model="gpt-4",
api_key="key-1",
exclude_entity_ids=["team-x"],
)
assert params == [
"2024-01-01",
"2024-01-31",
"team-abc",
"team-x",
"gpt-4",
"key-1",
]
assert '"team_id" = $3' in sql
assert '"team_id" NOT IN ($4)' in sql
assert "model = $5" in sql
assert "api_key = $6" in sql
def test_timezone_offset_positive(self):
"""Positive timezone offset (west of UTC) should extend end_date by 1 day."""
sql, params = _build_aggregated_sql_query(
table_name="litellm_dailyteamspend",
entity_id_field="team_id",
entity_id=None,
start_date="2024-01-01",
end_date="2024-01-31",
model=None,
api_key=None,
timezone_offset_minutes=480, # PST
)
assert params == ["2024-01-01", "2024-02-01"]
def test_timezone_offset_negative(self):
"""Negative timezone offset (east of UTC) should extend start_date by -1 day."""
sql, params = _build_aggregated_sql_query(
table_name="litellm_dailyteamspend",
entity_id_field="team_id",
entity_id=None,
start_date="2024-01-01",
end_date="2024-01-31",
model=None,
api_key=None,
timezone_offset_minutes=-330, # IST
)
assert params == ["2023-12-31", "2024-01-31"]
def test_unknown_table_name_raises(self):
"""Unknown table name should raise ValueError."""
with pytest.raises(ValueError, match="Unknown table name"):
_build_aggregated_sql_query(
table_name="nonexistent_table",
entity_id_field="team_id",
entity_id=None,
start_date="2024-01-01",
end_date="2024-01-31",
model=None,
api_key=None,
)
def test_all_table_names_supported(self):
"""All known table names should produce valid queries."""
tables = [
("litellm_dailyuserspend", "user_id", "LiteLLM_DailyUserSpend"),
("litellm_dailyteamspend", "team_id", "LiteLLM_DailyTeamSpend"),
(
"litellm_dailyorganizationspend",
"organization_id",
"LiteLLM_DailyOrganizationSpend",
),
("litellm_dailyenduserspend", "end_user_id", "LiteLLM_DailyEndUserSpend"),
("litellm_dailyagentspend", "agent_id", "LiteLLM_DailyAgentSpend"),
("litellm_dailytagspend", "tag", "LiteLLM_DailyTagSpend"),
]
for table_name, entity_field, pg_table in tables:
sql, params = _build_aggregated_sql_query(
table_name=table_name,
entity_id_field=entity_field,
entity_id=None,
start_date="2024-01-01",
end_date="2024-01-31",
model=None,
api_key=None,
)
assert f'FROM "{pg_table}"' in sql
assert "GROUP BY" not in sql