fix: cap spend_log_transactions queue to prevent unbounded memory growth

The PrismaClient.spend_log_transactions list is unbounded — every request
appends a deepcopy'd spend-log payload (~2-5KB). When the DB is unreachable
or flush can't keep up, this list grows without limit, causing workers to
crash with MemoryError.

Fix: Add MAX_SPEND_LOG_TRANSACTIONS cap (default 10000, ~20-50MB max).
When the queue is full, drop the oldest 10% of entries and log a warning.
This prevents OOM while preserving the most recent spend data.

Configurable via MAX_SPEND_LOG_TRANSACTIONS env var.

Co-authored-by: Ishaan Jaff <ishaan-jaff@users.noreply.github.com>
This commit is contained in:
Cursor Agent 2026-02-27 05:17:34 +00:00
parent 285c4a3c17
commit d931577678
2 changed files with 25 additions and 4 deletions

View file

@ -1193,6 +1193,11 @@ SPEND_LOG_RUN_LOOPS = int(os.getenv("SPEND_LOG_RUN_LOOPS", 500))
SPEND_LOG_CLEANUP_BATCH_SIZE = int(os.getenv("SPEND_LOG_CLEANUP_BATCH_SIZE", 1000))
SPEND_LOG_QUEUE_SIZE_THRESHOLD = int(os.getenv("SPEND_LOG_QUEUE_SIZE_THRESHOLD", 100))
SPEND_LOG_QUEUE_POLL_INTERVAL = float(os.getenv("SPEND_LOG_QUEUE_POLL_INTERVAL", 2.0))
# MEMORY LEAK FIX: Maximum number of spend log entries to hold in memory.
# When the queue exceeds this cap (e.g., DB is unreachable and flush fails),
# the oldest entries are dropped to prevent unbounded memory growth.
# Each spend log payload is ~2-5KB after deepcopy; 10000 entries ≈ 20-50MB.
MAX_SPEND_LOG_TRANSACTIONS = int(os.getenv("MAX_SPEND_LOG_TRANSACTIONS", 10000))
DEFAULT_CRON_JOB_LOCK_TTL_SECONDS = int(
os.getenv("DEFAULT_CRON_JOB_LOCK_TTL_SECONDS", 60)
) # 1 minute

View file

@ -422,16 +422,32 @@ class DBSpendUpdateWriter:
prisma_client: Optional[PrismaClient] = None,
spend_logs_url: Optional[str] = os.getenv("SPEND_LOGS_URL"),
) -> Optional[PrismaClient]:
from litellm.constants import MAX_SPEND_LOG_TRANSACTIONS
verbose_proxy_logger.debug(
"Writing spend log to db - request_id: {}, spend: {}".format(
payload.get("request_id"), payload.get("spend")
)
)
if prisma_client is not None and spend_logs_url is not None:
async with prisma_client._spend_log_transactions_lock:
prisma_client.spend_log_transactions.append(payload)
elif prisma_client is not None:
if prisma_client is not None:
async with prisma_client._spend_log_transactions_lock:
queue_len = len(prisma_client.spend_log_transactions)
if queue_len >= MAX_SPEND_LOG_TRANSACTIONS:
# MEMORY LEAK FIX: Drop oldest entries to prevent unbounded growth.
# This happens when the DB is unreachable or flush can't keep up.
drop_count = max(1, MAX_SPEND_LOG_TRANSACTIONS // 10)
prisma_client.spend_log_transactions = (
prisma_client.spend_log_transactions[drop_count:]
)
verbose_proxy_logger.warning(
"spend_log_transactions queue at capacity (%d). "
"Dropped oldest %d entries to prevent memory leak. "
"This usually means the database is unreachable or "
"flush cannot keep up with request volume. "
"Adjust MAX_SPEND_LOG_TRANSACTIONS env var to change the cap.",
queue_len,
drop_count,
)
prisma_client.spend_log_transactions.append(payload)
else:
verbose_proxy_logger.debug(