From d931577678e4cae589988269866026625a7434f5 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Fri, 27 Feb 2026 05:17:34 +0000 Subject: [PATCH] fix: cap spend_log_transactions queue to prevent unbounded memory growth MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The PrismaClient.spend_log_transactions list is unbounded — every request appends a deepcopy'd spend-log payload (~2-5KB). When the DB is unreachable or flush can't keep up, this list grows without limit, causing workers to crash with MemoryError. Fix: Add MAX_SPEND_LOG_TRANSACTIONS cap (default 10000, ~20-50MB max). When the queue is full, drop the oldest 10% of entries and log a warning. This prevents OOM while preserving the most recent spend data. Configurable via MAX_SPEND_LOG_TRANSACTIONS env var. Co-authored-by: Ishaan Jaff --- litellm/constants.py | 5 +++++ litellm/proxy/db/db_spend_update_writer.py | 24 ++++++++++++++++++---- 2 files changed, 25 insertions(+), 4 deletions(-) diff --git a/litellm/constants.py b/litellm/constants.py index dd4feab0c91..2e8f3f94f5a 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -1193,6 +1193,11 @@ SPEND_LOG_RUN_LOOPS = int(os.getenv("SPEND_LOG_RUN_LOOPS", 500)) SPEND_LOG_CLEANUP_BATCH_SIZE = int(os.getenv("SPEND_LOG_CLEANUP_BATCH_SIZE", 1000)) SPEND_LOG_QUEUE_SIZE_THRESHOLD = int(os.getenv("SPEND_LOG_QUEUE_SIZE_THRESHOLD", 100)) SPEND_LOG_QUEUE_POLL_INTERVAL = float(os.getenv("SPEND_LOG_QUEUE_POLL_INTERVAL", 2.0)) +# MEMORY LEAK FIX: Maximum number of spend log entries to hold in memory. +# When the queue exceeds this cap (e.g., DB is unreachable and flush fails), +# the oldest entries are dropped to prevent unbounded memory growth. +# Each spend log payload is ~2-5KB after deepcopy; 10000 entries ≈ 20-50MB. +MAX_SPEND_LOG_TRANSACTIONS = int(os.getenv("MAX_SPEND_LOG_TRANSACTIONS", 10000)) DEFAULT_CRON_JOB_LOCK_TTL_SECONDS = int( os.getenv("DEFAULT_CRON_JOB_LOCK_TTL_SECONDS", 60) ) # 1 minute diff --git a/litellm/proxy/db/db_spend_update_writer.py b/litellm/proxy/db/db_spend_update_writer.py index 429e56c805b..49501f93fe0 100644 --- a/litellm/proxy/db/db_spend_update_writer.py +++ b/litellm/proxy/db/db_spend_update_writer.py @@ -422,16 +422,32 @@ class DBSpendUpdateWriter: prisma_client: Optional[PrismaClient] = None, spend_logs_url: Optional[str] = os.getenv("SPEND_LOGS_URL"), ) -> Optional[PrismaClient]: + from litellm.constants import MAX_SPEND_LOG_TRANSACTIONS + verbose_proxy_logger.debug( "Writing spend log to db - request_id: {}, spend: {}".format( payload.get("request_id"), payload.get("spend") ) ) - if prisma_client is not None and spend_logs_url is not None: - async with prisma_client._spend_log_transactions_lock: - prisma_client.spend_log_transactions.append(payload) - elif prisma_client is not None: + if prisma_client is not None: async with prisma_client._spend_log_transactions_lock: + queue_len = len(prisma_client.spend_log_transactions) + if queue_len >= MAX_SPEND_LOG_TRANSACTIONS: + # MEMORY LEAK FIX: Drop oldest entries to prevent unbounded growth. + # This happens when the DB is unreachable or flush can't keep up. + drop_count = max(1, MAX_SPEND_LOG_TRANSACTIONS // 10) + prisma_client.spend_log_transactions = ( + prisma_client.spend_log_transactions[drop_count:] + ) + verbose_proxy_logger.warning( + "spend_log_transactions queue at capacity (%d). " + "Dropped oldest %d entries to prevent memory leak. " + "This usually means the database is unreachable or " + "flush cannot keep up with request volume. " + "Adjust MAX_SPEND_LOG_TRANSACTIONS env var to change the cap.", + queue_len, + drop_count, + ) prisma_client.spend_log_transactions.append(payload) else: verbose_proxy_logger.debug(