mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-23 00:41:40 +00:00
fix(proxy): report total_api_keys so exact-limit key sets are not treated as truncated
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
aec36a2bac
commit
87894f2e7f
9 changed files with 126 additions and 45 deletions
|
|
@ -3072,6 +3072,18 @@
|
|||
"title": "Page",
|
||||
"type": "integer"
|
||||
},
|
||||
"total_api_keys": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "integer"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"description": "Distinct API keys matching the filters. When this exceeds api_key_limit, the per-key lists are truncated to the highest-spend keys.",
|
||||
"title": "Total Api Keys"
|
||||
},
|
||||
"total_api_requests": {
|
||||
"default": 0,
|
||||
"title": "Total Api Requests",
|
||||
|
|
|
|||
|
|
@ -155,6 +155,7 @@ class _GroupingSetsRow(SimpleNamespace):
|
|||
mcp_namespaced_tool_name: str | None
|
||||
endpoint: str | None
|
||||
group_level: int
|
||||
distinct_api_keys: int | None
|
||||
spend: float | None
|
||||
prompt_tokens: int | None
|
||||
completion_tokens: int | None
|
||||
|
|
@ -800,7 +801,8 @@ def _build_aggregated_sql_query(
|
|||
(GROUPING(date) << 6) | {_API_KEY_ROLLED_UP_BIT}
|
||||
| GROUPING(model, {_MODEL_GROUP_EXPR},
|
||||
custom_llm_provider, mcp_namespaced_tool_name,
|
||||
endpoint) AS group_level,{metric_select}
|
||||
endpoint) AS group_level,
|
||||
NULL::bigint AS distinct_api_keys,{metric_select}
|
||||
FROM "{pg_table}"
|
||||
WHERE {where_clause}
|
||||
GROUP BY GROUPING SETS (
|
||||
|
|
@ -814,7 +816,7 @@ def _build_aggregated_sql_query(
|
|||
))
|
||||
UNION ALL
|
||||
(WITH top_api_keys AS (
|
||||
SELECT api_key
|
||||
SELECT api_key, COUNT(*) OVER () AS distinct_api_keys
|
||||
FROM "{pg_table}"
|
||||
WHERE {where_clause} AND api_key <> {sentinel_param}
|
||||
GROUP BY api_key
|
||||
|
|
@ -831,9 +833,10 @@ def _build_aggregated_sql_query(
|
|||
endpoint,
|
||||
GROUPING(date, api_key, model, {_MODEL_GROUP_EXPR},
|
||||
custom_llm_provider, mcp_namespaced_tool_name,
|
||||
endpoint) AS group_level,{metric_select}
|
||||
FROM "{pg_table}"
|
||||
WHERE {where_clause} AND api_key IN (SELECT api_key FROM top_api_keys)
|
||||
endpoint) AS group_level,
|
||||
MAX(top_api_keys.distinct_api_keys) AS distinct_api_keys,{metric_select}
|
||||
FROM "{pg_table}" JOIN top_api_keys USING (api_key)
|
||||
WHERE {where_clause}
|
||||
GROUP BY GROUPING SETS (
|
||||
(date, api_key),
|
||||
(date, model, api_key),
|
||||
|
|
@ -1398,6 +1401,7 @@ async def get_daily_activity_aggregated(
|
|||
)
|
||||
|
||||
records: Final = [_GroupingSetsRow(**row) for row in (raw_rows or ())]
|
||||
total_api_keys: Final = next((r.distinct_api_keys for r in records if r.distinct_api_keys is not None), 0)
|
||||
|
||||
# The grouping-sets dispatcher places each row directly in its bucket
|
||||
# using the row's GROUPING() bitmask. No Python-side summing needed.
|
||||
|
|
@ -1452,6 +1456,7 @@ async def get_daily_activity_aggregated(
|
|||
total_pages=1,
|
||||
has_more=False,
|
||||
api_key_limit=USAGE_TOP_API_KEYS_LIMIT,
|
||||
total_api_keys=total_api_keys,
|
||||
),
|
||||
)
|
||||
|
||||
|
|
|
|||
|
|
@ -105,6 +105,11 @@ class DailySpendMetadata(BaseModel):
|
|||
description="When set, api_keys and every api_key_breakdown list at most this many keys, "
|
||||
"ranked by spend. Totals and the model, provider, mcp and endpoint rollups still cover every key.",
|
||||
)
|
||||
total_api_keys: int | None = Field(
|
||||
default=None,
|
||||
description="Distinct API keys matching the filters. When this exceeds api_key_limit, the per-key "
|
||||
"lists are truncated to the highest-spend keys.",
|
||||
)
|
||||
|
||||
|
||||
class SpendAnalyticsPaginatedResponse(BaseModel):
|
||||
|
|
|
|||
|
|
@ -175,6 +175,7 @@ async def test_get_daily_activity_aggregated_with_endpoint_breakdown():
|
|||
"endpoint": "/v1/chat/completions",
|
||||
"api_key": None,
|
||||
"group_level": 62,
|
||||
"distinct_api_keys": None,
|
||||
"spend": 15.0,
|
||||
"prompt_tokens": 150,
|
||||
"completion_tokens": 75,
|
||||
|
|
@ -187,6 +188,7 @@ async def test_get_daily_activity_aggregated_with_endpoint_breakdown():
|
|||
"endpoint": "/v1/embeddings",
|
||||
"api_key": None,
|
||||
"group_level": 62,
|
||||
"distinct_api_keys": None,
|
||||
"spend": 3.0,
|
||||
"prompt_tokens": 30,
|
||||
"completion_tokens": 0,
|
||||
|
|
@ -200,6 +202,7 @@ async def test_get_daily_activity_aggregated_with_endpoint_breakdown():
|
|||
"endpoint": None,
|
||||
"api_key": None,
|
||||
"group_level": 63,
|
||||
"distinct_api_keys": None,
|
||||
"spend": 18.0,
|
||||
"prompt_tokens": 180,
|
||||
"completion_tokens": 75,
|
||||
|
|
@ -213,6 +216,7 @@ async def test_get_daily_activity_aggregated_with_endpoint_breakdown():
|
|||
"endpoint": None,
|
||||
"api_key": None,
|
||||
"group_level": 127,
|
||||
"distinct_api_keys": None,
|
||||
"spend": 18.0,
|
||||
"prompt_tokens": 180,
|
||||
"completion_tokens": 75,
|
||||
|
|
@ -226,6 +230,7 @@ async def test_get_daily_activity_aggregated_with_endpoint_breakdown():
|
|||
"endpoint": "/v1/chat/completions",
|
||||
"api_key": "key-1",
|
||||
"group_level": 30,
|
||||
"distinct_api_keys": 2,
|
||||
"spend": 15.0,
|
||||
"prompt_tokens": 150,
|
||||
"completion_tokens": 75,
|
||||
|
|
@ -238,6 +243,7 @@ async def test_get_daily_activity_aggregated_with_endpoint_breakdown():
|
|||
"endpoint": "/v1/embeddings",
|
||||
"api_key": "key-2",
|
||||
"group_level": 30,
|
||||
"distinct_api_keys": 2,
|
||||
"spend": 3.0,
|
||||
"prompt_tokens": 30,
|
||||
"completion_tokens": 0,
|
||||
|
|
@ -839,6 +845,7 @@ async def test_aggregated_activity_preserves_metadata_for_deleted_keys():
|
|||
"endpoint": "/v1/chat/completions",
|
||||
"api_key": None,
|
||||
"group_level": 62,
|
||||
"distinct_api_keys": None,
|
||||
"spend": 10.0,
|
||||
"prompt_tokens": 100,
|
||||
"completion_tokens": 50,
|
||||
|
|
@ -851,6 +858,7 @@ async def test_aggregated_activity_preserves_metadata_for_deleted_keys():
|
|||
"endpoint": "/v1/chat/completions",
|
||||
"api_key": "deleted-key-hash",
|
||||
"group_level": 30,
|
||||
"distinct_api_keys": 1,
|
||||
"spend": 10.0,
|
||||
"prompt_tokens": 100,
|
||||
"completion_tokens": 50,
|
||||
|
|
@ -1316,6 +1324,7 @@ async def test_get_daily_activity_aggregated_empty_result_set():
|
|||
"mcp_namespaced_tool_name": None,
|
||||
"endpoint": None,
|
||||
"group_level": 127,
|
||||
"distinct_api_keys": None,
|
||||
"spend": None,
|
||||
"prompt_tokens": None,
|
||||
"completion_tokens": None,
|
||||
|
|
@ -1499,6 +1508,7 @@ async def test_get_daily_activity_aggregated_bounds_api_key_rollups(
|
|||
assert result.metadata.total_spend == pytest.approx(key_spend + 1000.0)
|
||||
assert result.metadata.total_api_requests == n_keys
|
||||
assert result.metadata.api_key_limit == USAGE_TOP_API_KEYS_LIMIT
|
||||
assert result.metadata.total_api_keys == n_keys
|
||||
|
||||
expected_top: Final = {f"key-{i:03d}" for i in range(6, n_keys)} | {"key-004"}
|
||||
day: Final = result.results[0]
|
||||
|
|
@ -1560,6 +1570,7 @@ async def test_get_daily_activity_aggregated_explicit_api_key_filter_scopes_both
|
|||
)
|
||||
|
||||
assert result.metadata.total_spend == 2.0
|
||||
assert result.metadata.total_api_keys == 1
|
||||
day: Final = result.results[0]
|
||||
assert set(day.breakdown.api_keys) == {"key-1"}
|
||||
assert day.breakdown.api_keys["key-1"].metrics.spend == 2.0
|
||||
|
|
@ -1567,6 +1578,55 @@ async def test_get_daily_activity_aggregated_explicit_api_key_filter_scopes_both
|
|||
assert set(day.breakdown.models["gpt-5"].api_key_breakdown) == {"key-1"}
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_daily_activity_aggregated_reports_exact_limit_key_count_as_complete(
|
||||
_aggregated_postgresql: psycopg.Connection,
|
||||
):
|
||||
"""With exactly USAGE_TOP_API_KEYS_LIMIT keys nothing is dropped, and the
|
||||
response must say so: total_api_keys equals the limit rather than exceeding it."""
|
||||
rows: Final = [
|
||||
(
|
||||
f"row-{i:03d}",
|
||||
f"user-{i:03d}",
|
||||
"2026-06-01",
|
||||
f"key-{i:03d}",
|
||||
"gpt-5",
|
||||
"",
|
||||
"openai",
|
||||
"/v1/chat/completions",
|
||||
10,
|
||||
float(i + 1),
|
||||
1,
|
||||
1,
|
||||
)
|
||||
for i in range(USAGE_TOP_API_KEYS_LIMIT)
|
||||
]
|
||||
_seed_daily_user_spend(_aggregated_postgresql, rows)
|
||||
|
||||
row_counts: Final[list[int]] = [] # mutable-ok: out-param for the query_raw shim
|
||||
mock_prisma = MagicMock()
|
||||
mock_prisma.db = MagicMock()
|
||||
mock_prisma.db.query_raw = _psycopg_query_raw(_aggregated_postgresql, row_counts)
|
||||
mock_prisma.db.litellm_verificationtoken.find_many = AsyncMock(return_value=[])
|
||||
mock_prisma.db.litellm_deletedverificationtoken.find_many = AsyncMock(return_value=[])
|
||||
|
||||
result = await get_daily_activity_aggregated(
|
||||
prisma_client=mock_prisma,
|
||||
table_name="litellm_dailyuserspend",
|
||||
entity_id_field="user_id",
|
||||
entity_id=None,
|
||||
entity_metadata_field=None,
|
||||
start_date="2026-06-01",
|
||||
end_date="2026-06-01",
|
||||
model=None,
|
||||
api_key=None,
|
||||
)
|
||||
|
||||
assert result.metadata.total_api_keys == USAGE_TOP_API_KEYS_LIMIT
|
||||
assert result.metadata.api_key_limit == USAGE_TOP_API_KEYS_LIMIT
|
||||
assert len(result.results[0].breakdown.api_keys) == USAGE_TOP_API_KEYS_LIMIT
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_daily_activity_aggregated_model_group_rollups_fall_back_to_model_name(
|
||||
_aggregated_postgresql: psycopg.Connection,
|
||||
|
|
@ -2427,10 +2487,10 @@ async def test_get_daily_activity_aggregated_with_entity_breakdown():
|
|||
"successful_requests": 0,
|
||||
}
|
||||
main_rows = [
|
||||
{**base, "date": None, "group_level": 127, "spend": 18.0},
|
||||
{**base, "date": "2024-01-01", "group_level": 63, "spend": 18.0},
|
||||
{**base, "date": "2024-01-01", "model": "gpt-4o", "group_level": 47, "spend": 18.0},
|
||||
{**base, "date": "2024-01-01", "api_key": "key-1", "group_level": 31, "spend": 12.0},
|
||||
{**base, "date": None, "group_level": 127, "distinct_api_keys": None, "spend": 18.0},
|
||||
{**base, "date": "2024-01-01", "group_level": 63, "distinct_api_keys": None, "spend": 18.0},
|
||||
{**base, "date": "2024-01-01", "model": "gpt-4o", "group_level": 47, "distinct_api_keys": None, "spend": 18.0},
|
||||
{**base, "date": "2024-01-01", "api_key": "key-1", "group_level": 31, "distinct_api_keys": 1, "spend": 12.0},
|
||||
]
|
||||
entity_base = {
|
||||
key: value
|
||||
|
|
|
|||
|
|
@ -664,7 +664,7 @@ const EntityUsage: React.FC<EntityUsageProps> = ({
|
|||
{ key: "endpoints", label: "Endpoint Activity", content: <EndpointUsage userSpendData={spendData} /> },
|
||||
];
|
||||
|
||||
const spendFetchState = { coversRange, cancelled, failed, apiKeyLimitReached: undefined };
|
||||
const spendFetchState = { coversRange, cancelled, failed, apiKeyTruncation: undefined };
|
||||
|
||||
return (
|
||||
<div style={{ width: "100%" }} className="relative">
|
||||
|
|
|
|||
|
|
@ -30,7 +30,7 @@ import { ActivityMetrics, processActivityData } from "@/components/activity_metr
|
|||
import CloudZeroExportModal from "@/components/cloudzero_export_modal";
|
||||
import UserDropdown from "@/components/common_components/UserDropdown";
|
||||
import EntityUsageExportModal from "@/components/EntityUsageExport";
|
||||
import { getApiKeyLimitReached, getExportBlockedReason } from "@/components/EntityUsageExport/exportBlockedReason";
|
||||
import { getApiKeyTruncation, getExportBlockedReason } from "@/components/EntityUsageExport/exportBlockedReason";
|
||||
import KeyActivityPanel from "@/components/UsagePage/components/KeyActivityPanel";
|
||||
import { Team } from "@/components/key_team_helpers/key_list";
|
||||
import {
|
||||
|
|
@ -256,7 +256,10 @@ const UsagePage: React.FC<UsagePageProps> = ({ teams, organizations }) => {
|
|||
coversRange: activeAggregated !== null || paginatedResult.coversRange,
|
||||
cancelled: paginatedResult.cancelled,
|
||||
failed: paginatedResult.failed,
|
||||
apiKeyLimitReached: getApiKeyLimitReached(userSpendData.results, userSpendData.metadata?.api_key_limit),
|
||||
apiKeyTruncation: getApiKeyTruncation(
|
||||
userSpendData.metadata?.api_key_limit,
|
||||
userSpendData.metadata?.total_api_keys,
|
||||
),
|
||||
};
|
||||
const exportBlockedReason = getExportBlockedReason(spendFetchState);
|
||||
|
||||
|
|
|
|||
|
|
@ -1,24 +1,15 @@
|
|||
import { describe, expect, it } from "vitest";
|
||||
|
||||
import type { DailyData } from "@/components/UsagePage/types";
|
||||
|
||||
import { getApiKeyLimitReached, getExportBlockedReason, type UsageFetchState } from "./exportBlockedReason";
|
||||
import { getApiKeyTruncation, getExportBlockedReason, type UsageFetchState } from "./exportBlockedReason";
|
||||
|
||||
const state = (overrides: Partial<UsageFetchState> = {}): UsageFetchState => ({
|
||||
coversRange: true,
|
||||
cancelled: false,
|
||||
failed: false,
|
||||
apiKeyLimitReached: undefined,
|
||||
apiKeyTruncation: undefined,
|
||||
...overrides,
|
||||
});
|
||||
|
||||
const dayWithKeys = (date: string, ...keys: string[]): DailyData =>
|
||||
({
|
||||
date,
|
||||
metrics: {},
|
||||
breakdown: { api_keys: Object.fromEntries(keys.map((k) => [k, { metrics: {}, metadata: {} }])) },
|
||||
}) as unknown as DailyData;
|
||||
|
||||
describe("getExportBlockedReason", () => {
|
||||
it("lets the export through once the data on screen covers the range", () => {
|
||||
expect(getExportBlockedReason(state())).toBeUndefined();
|
||||
|
|
@ -42,28 +33,26 @@ describe("getExportBlockedReason", () => {
|
|||
expect(reason).not.toMatch(/stopped/i);
|
||||
});
|
||||
|
||||
it("blocks when the aggregated endpoint hit its key cap, since a per-team CSV would miss keys", () => {
|
||||
const reason = getExportBlockedReason(state({ apiKeyLimitReached: 100 }));
|
||||
it("blocks when the aggregated endpoint dropped keys, since a per-team CSV would miss them", () => {
|
||||
const reason = getExportBlockedReason(state({ apiKeyTruncation: { limit: 100, total: 3000 } }));
|
||||
|
||||
expect(reason).toMatch(/100 highest-spend keys/);
|
||||
expect(reason).toMatch(/100 highest-spend keys of 3000/);
|
||||
expect(reason).toMatch(/USAGE_TOP_API_KEYS_LIMIT/);
|
||||
});
|
||||
});
|
||||
|
||||
describe("getApiKeyLimitReached", () => {
|
||||
it("reports the cap once the distinct keys across every day reach it", () => {
|
||||
const results = [dayWithKeys("2026-06-01", "key-1", "key-2"), dayWithKeys("2026-06-02", "key-2", "key-3")];
|
||||
|
||||
expect(getApiKeyLimitReached(results, 3)).toBe(3);
|
||||
describe("getApiKeyTruncation", () => {
|
||||
it("reports truncation once the proxy saw more keys than it returned", () => {
|
||||
expect(getApiKeyTruncation(100, 101)).toEqual({ limit: 100, total: 101 });
|
||||
});
|
||||
|
||||
it("stays quiet while fewer keys than the cap came back, which means every key is on screen", () => {
|
||||
const results = [dayWithKeys("2026-06-01", "key-1", "key-2"), dayWithKeys("2026-06-02", "key-2")];
|
||||
|
||||
expect(getApiKeyLimitReached(results, 3)).toBeUndefined();
|
||||
it("stays quiet when exactly the cap exists, since every key is on screen", () => {
|
||||
expect(getApiKeyTruncation(100, 100)).toBeUndefined();
|
||||
expect(getApiKeyTruncation(100, 7)).toBeUndefined();
|
||||
});
|
||||
|
||||
it("stays quiet when the response carries no cap, as the paginated fallback does", () => {
|
||||
expect(getApiKeyLimitReached([dayWithKeys("2026-06-01", "key-1")], undefined)).toBeUndefined();
|
||||
expect(getApiKeyTruncation(undefined, undefined)).toBeUndefined();
|
||||
expect(getApiKeyTruncation(100, null)).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -1,29 +1,31 @@
|
|||
import type { DailyData } from "@/components/UsagePage/types";
|
||||
export interface ApiKeyTruncation {
|
||||
limit: number;
|
||||
total: number;
|
||||
}
|
||||
|
||||
export interface UsageFetchState {
|
||||
coversRange: boolean;
|
||||
cancelled: boolean;
|
||||
failed: boolean;
|
||||
apiKeyLimitReached: number | undefined;
|
||||
apiKeyTruncation: ApiKeyTruncation | undefined;
|
||||
}
|
||||
|
||||
export const getApiKeyLimitReached = (results: DailyData[], apiKeyLimit: unknown): number | undefined => {
|
||||
if (typeof apiKeyLimit !== "number") return undefined;
|
||||
const keys = new Set(results.flatMap((day) => Object.keys(day.breakdown.api_keys ?? {})));
|
||||
return keys.size >= apiKeyLimit ? apiKeyLimit : undefined;
|
||||
export const getApiKeyTruncation = (apiKeyLimit: unknown, totalApiKeys: unknown): ApiKeyTruncation | undefined => {
|
||||
if (typeof apiKeyLimit !== "number" || typeof totalApiKeys !== "number") return undefined;
|
||||
return totalApiKeys > apiKeyLimit ? { limit: apiKeyLimit, total: totalApiKeys } : undefined;
|
||||
};
|
||||
|
||||
export const getExportBlockedReason = ({
|
||||
coversRange,
|
||||
cancelled,
|
||||
failed,
|
||||
apiKeyLimitReached,
|
||||
apiKeyTruncation,
|
||||
}: UsageFetchState): string | undefined => {
|
||||
if (failed) return "Some spend data failed to load, so an export would under-report. Reload the page to try again.";
|
||||
if (cancelled)
|
||||
return "Loading was stopped before the whole range arrived, so an export would under-report. Reload the page to load it all.";
|
||||
if (!coversRange) return "Spend data is still loading, so an export would under-report. Wait for it to finish.";
|
||||
if (apiKeyLimitReached !== undefined)
|
||||
return `Only the ${apiKeyLimitReached} highest-spend keys were loaded, so a per-team export would under-report. Raise USAGE_TOP_API_KEYS_LIMIT on the proxy to load more keys.`;
|
||||
if (apiKeyTruncation !== undefined)
|
||||
return `Only the ${apiKeyTruncation.limit} highest-spend keys of ${apiKeyTruncation.total} were loaded, so a per-team export would under-report. Raise USAGE_TOP_API_KEYS_LIMIT on the proxy to load more keys.`;
|
||||
return undefined;
|
||||
};
|
||||
|
|
|
|||
5
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
5
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
|
|
@ -27503,6 +27503,11 @@ export interface components {
|
|||
* @default 1
|
||||
*/
|
||||
page: number;
|
||||
/**
|
||||
* Total Api Keys
|
||||
* @description Distinct API keys matching the filters. When this exceeds api_key_limit, the per-key lists are truncated to the highest-spend keys.
|
||||
*/
|
||||
total_api_keys?: number | null;
|
||||
/**
|
||||
* Total Api Requests
|
||||
* @default 0
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue