mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
fix: cap spend_log_transactions queue to prevent unbounded memory growth
The PrismaClient.spend_log_transactions list is unbounded — every request appends a deepcopy'd spend-log payload (~2-5KB). When the DB is unreachable or flush can't keep up, this list grows without limit, causing workers to crash with MemoryError. Fix: Add MAX_SPEND_LOG_TRANSACTIONS cap (default 10000, ~20-50MB max). When the queue is full, drop the oldest 10% of entries and log a warning. This prevents OOM while preserving the most recent spend data. Configurable via MAX_SPEND_LOG_TRANSACTIONS env var. Co-authored-by: Ishaan Jaff <ishaan-jaff@users.noreply.github.com>
This commit is contained in:
parent
285c4a3c17
commit
d931577678
2 changed files with 25 additions and 4 deletions
|
|
@ -1193,6 +1193,11 @@ SPEND_LOG_RUN_LOOPS = int(os.getenv("SPEND_LOG_RUN_LOOPS", 500))
|
|||
SPEND_LOG_CLEANUP_BATCH_SIZE = int(os.getenv("SPEND_LOG_CLEANUP_BATCH_SIZE", 1000))
|
||||
SPEND_LOG_QUEUE_SIZE_THRESHOLD = int(os.getenv("SPEND_LOG_QUEUE_SIZE_THRESHOLD", 100))
|
||||
SPEND_LOG_QUEUE_POLL_INTERVAL = float(os.getenv("SPEND_LOG_QUEUE_POLL_INTERVAL", 2.0))
|
||||
# MEMORY LEAK FIX: Maximum number of spend log entries to hold in memory.
|
||||
# When the queue exceeds this cap (e.g., DB is unreachable and flush fails),
|
||||
# the oldest entries are dropped to prevent unbounded memory growth.
|
||||
# Each spend log payload is ~2-5KB after deepcopy; 10000 entries ≈ 20-50MB.
|
||||
MAX_SPEND_LOG_TRANSACTIONS = int(os.getenv("MAX_SPEND_LOG_TRANSACTIONS", 10000))
|
||||
DEFAULT_CRON_JOB_LOCK_TTL_SECONDS = int(
|
||||
os.getenv("DEFAULT_CRON_JOB_LOCK_TTL_SECONDS", 60)
|
||||
) # 1 minute
|
||||
|
|
|
|||
|
|
@ -422,16 +422,32 @@ class DBSpendUpdateWriter:
|
|||
prisma_client: Optional[PrismaClient] = None,
|
||||
spend_logs_url: Optional[str] = os.getenv("SPEND_LOGS_URL"),
|
||||
) -> Optional[PrismaClient]:
|
||||
from litellm.constants import MAX_SPEND_LOG_TRANSACTIONS
|
||||
|
||||
verbose_proxy_logger.debug(
|
||||
"Writing spend log to db - request_id: {}, spend: {}".format(
|
||||
payload.get("request_id"), payload.get("spend")
|
||||
)
|
||||
)
|
||||
if prisma_client is not None and spend_logs_url is not None:
|
||||
async with prisma_client._spend_log_transactions_lock:
|
||||
prisma_client.spend_log_transactions.append(payload)
|
||||
elif prisma_client is not None:
|
||||
if prisma_client is not None:
|
||||
async with prisma_client._spend_log_transactions_lock:
|
||||
queue_len = len(prisma_client.spend_log_transactions)
|
||||
if queue_len >= MAX_SPEND_LOG_TRANSACTIONS:
|
||||
# MEMORY LEAK FIX: Drop oldest entries to prevent unbounded growth.
|
||||
# This happens when the DB is unreachable or flush can't keep up.
|
||||
drop_count = max(1, MAX_SPEND_LOG_TRANSACTIONS // 10)
|
||||
prisma_client.spend_log_transactions = (
|
||||
prisma_client.spend_log_transactions[drop_count:]
|
||||
)
|
||||
verbose_proxy_logger.warning(
|
||||
"spend_log_transactions queue at capacity (%d). "
|
||||
"Dropped oldest %d entries to prevent memory leak. "
|
||||
"This usually means the database is unreachable or "
|
||||
"flush cannot keep up with request volume. "
|
||||
"Adjust MAX_SPEND_LOG_TRANSACTIONS env var to change the cap.",
|
||||
queue_len,
|
||||
drop_count,
|
||||
)
|
||||
prisma_client.spend_log_transactions.append(payload)
|
||||
else:
|
||||
verbose_proxy_logger.debug(
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue