From d919b1048e9bac832befb58a205c17cd23713006 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Wed, 25 Feb 2026 23:05:12 +0000 Subject: [PATCH] fix: reduce memory retention after traffic spikes Two-pronged fix for the reported memory leak where RSS grows from ~1.9 GiB to ~3.5 GiB under load (1000-1500 concurrent users) and never returns to baseline: 1. Logging._cleanup_after_logging(): Clears large payload fields (input, original_response, raw_request_typed_dict) and streaming chunk lists from the Logging object after all success/failure callbacks have consumed them. This prevents per-request data (which can be kilobytes per request) from being held in memory by async task references. 2. _periodic_memory_cleanup(): A scheduled background job (every 60s by default, configurable via LITELLM_MEMORY_CLEANUP_INTERVAL) that runs gc.collect() + malloc_trim(0) on Linux. Python's pymalloc allocator does not return freed memory arenas to the OS by default; malloc_trim forces glibc to release unused heap pages, allowing RSS to shrink after traffic spikes subside. Together these ensure that: - Per-request memory is released promptly after logging completes - The OS reclaims freed heap memory periodically, so RSS drops during idle periods instead of staying at the peak level indefinitely Co-authored-by: Ishaan Jaff --- AGENTS.md | 23 ++++++++- litellm/litellm_core_utils/litellm_logging.py | 42 +++++++++++++++++ litellm/proxy/common_utils/memory_utils.py | 47 +++++++++++++++++++ litellm/proxy/proxy_server.py | 17 +++++++ 4 files changed, 128 insertions(+), 1 deletion(-) create mode 100644 litellm/proxy/common_utils/memory_utils.py diff --git a/AGENTS.md b/AGENTS.md index bfd44304d55..422dfffd450 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -189,4 +189,25 @@ When opening issues or pull requests, follow these templates: - Check similar provider implementations - Ensure comprehensive test coverage - Update documentation appropriately -- Consider backward compatibility impact \ No newline at end of file +- Consider backward compatibility impact + +## Cursor Cloud specific instructions + +### Environment setup +- Python 3.12 with `poetry install --with dev,proxy-dev --extras proxy` +- Install `psycopg-binary` and `psutil` separately: `poetry run pip install psycopg-binary psutil` +- Lint: `cd litellm && poetry run ruff check .` +- Unit tests: `poetry run pytest tests/test_litellm/ -x -n 4` (see `Makefile` for specific test groups) +- Memory/load tests: `poetry run pytest tests/load_tests/test_linear_memory_growth.py::test_memory_baseline_1k -v` (run individually, not all at once) + +### Running the proxy server locally +- `poetry run litellm --model openai/gpt-4o` (requires `OPENAI_API_KEY`) +- Proxy listens on port 4000 by default +- No PostgreSQL or Redis required for basic SDK/proxy testing with mock endpoints + +### Memory leak testing +- Existing tests in `tests/load_tests/` use a local mock OpenAI server (no API keys needed) +- `test_linear_memory_growth.py` tests are designed to run **individually** (memory baselines drift when combined) +- `_periodic_memory_cleanup()` in `litellm/proxy/common_utils/memory_utils.py` calls `gc.collect()` + `malloc_trim()` on Linux to return freed memory to the OS +- `LITELLM_MEMORY_CLEANUP_INTERVAL` env var controls cleanup frequency (default 60s) +- `PYTHON_GC_THRESHOLD` env var configures GC thresholds (format: `gen0,gen1,gen2`) \ No newline at end of file diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index 0601e7e8455..e840e7df614 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -2368,6 +2368,9 @@ class Logging(LiteLLMLoggingBaseClass): str(e) ), ) + finally: + # Release large data structures now that all callbacks have consumed them + self._cleanup_after_logging() async def async_success_handler( # noqa: PLR0915 self, result=None, start_time=None, end_time=None, cache_hit=None, **kwargs @@ -2713,6 +2716,45 @@ class Logging(LiteLLMLoggingBaseClass): self._handle_callback_failure(callback=callback) pass + # Release large data structures now that all callbacks have consumed them + self._cleanup_after_logging() + + def _cleanup_after_logging(self): + """ + Release large data structures after all logging callbacks have completed. + + This prevents memory from accumulating when many requests are processed, + especially under high concurrency. The Logging object may still be + referenced by async tasks, but its large payload fields are no longer + needed after callbacks have consumed them. + + Fields intentionally kept: + - ``response_cost``, ``standard_logging_object`` – may be read after + callbacks (e.g. for response headers). + - ``async_complete_streaming_response``, ``complete_streaming_response`` + – used as guard flags to prevent double-processing on repeated calls. + """ + # Clear streaming chunks (can be very large for long responses) + if hasattr(self, "streaming_chunks"): + self.streaming_chunks.clear() + if hasattr(self, "sync_streaming_chunks"): + self.sync_streaming_chunks.clear() + + # Clear large payload-only fields from model_call_details. + # These hold the raw request input and response text which can be + # substantial (kilobytes to megabytes per request). + _large_keys = ( + "input", + "original_response", + "complete_response", + "raw_request_typed_dict", + ) + for key in _large_keys: + self.model_call_details.pop(key, None) + + # Clear messages reference (the full request prompt) + self.messages = None + def _handle_callback_failure(self, callback: Any): """ Handle callback logging failures by incrementing Prometheus metrics. diff --git a/litellm/proxy/common_utils/memory_utils.py b/litellm/proxy/common_utils/memory_utils.py new file mode 100644 index 00000000000..a19ff3437bd --- /dev/null +++ b/litellm/proxy/common_utils/memory_utils.py @@ -0,0 +1,47 @@ +""" +Periodic memory cleanup utilities for the LiteLLM proxy. + +Python's memory allocator (pymalloc) does not return freed memory arenas back +to the OS by default. After handling a burst of traffic, process RSS stays +elevated even though the objects have been garbage-collected. This module +provides a lightweight scheduled job that: + +1. Runs a full GC collection cycle +2. Calls ``malloc_trim(0)`` on Linux (glibc) to return unused heap pages to + the OS + +This is especially important for long-running proxy deployments where memory +grows during load tests / peak traffic and never shrinks back. + +Environment variables: + LITELLM_MEMORY_CLEANUP_INTERVAL – seconds between cleanup runs (default 60) +""" + +import ctypes +import gc +import sys + +from litellm._logging import verbose_proxy_logger + +_libc = None +_malloc_trim_available = False + +if sys.platform == "linux": + try: + _libc = ctypes.CDLL("libc.so.6") + _malloc_trim_available = hasattr(_libc, "malloc_trim") + except OSError: + pass + + +def _periodic_memory_cleanup() -> None: + """Run GC and release freed memory back to the OS.""" + gc.collect() + + if _malloc_trim_available and _libc is not None: + try: + _libc.malloc_trim(0) + except Exception as exc: + verbose_proxy_logger.debug( + "malloc_trim failed (non-critical): %s", exc + ) diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 82cfd455be6..d51a9ab2324 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -5667,6 +5667,23 @@ class ProxyStartupEvent: misfire_grace_time=APSCHEDULER_MISFIRE_GRACE_TIME, ) + ### PERIODIC MEMORY CLEANUP ### + # Return freed memory to the OS. Python's allocator holds freed arenas + # indefinitely; malloc_trim() on Linux releases them, preventing RSS + # from staying elevated after traffic spikes. + from litellm.proxy.common_utils.memory_utils import ( + _periodic_memory_cleanup, + ) + + scheduler.add_job( + _periodic_memory_cleanup, + "interval", + seconds=int(os.getenv("LITELLM_MEMORY_CLEANUP_INTERVAL", 60)), + id="memory_cleanup_job", + replace_existing=True, + misfire_grace_time=APSCHEDULER_MISFIRE_GRACE_TIME, + ) + ### MONITOR SPEND LOGS QUEUE (queue-size-based job) ### if general_settings.get("disable_spend_logs", False) is False: from litellm.proxy.utils import _monitor_spend_logs_queue