mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
fix: reduce memory retention after traffic spikes
Two-pronged fix for the reported memory leak where RSS grows from ~1.9 GiB to ~3.5 GiB under load (1000-1500 concurrent users) and never returns to baseline: 1. Logging._cleanup_after_logging(): Clears large payload fields (input, original_response, raw_request_typed_dict) and streaming chunk lists from the Logging object after all success/failure callbacks have consumed them. This prevents per-request data (which can be kilobytes per request) from being held in memory by async task references. 2. _periodic_memory_cleanup(): A scheduled background job (every 60s by default, configurable via LITELLM_MEMORY_CLEANUP_INTERVAL) that runs gc.collect() + malloc_trim(0) on Linux. Python's pymalloc allocator does not return freed memory arenas to the OS by default; malloc_trim forces glibc to release unused heap pages, allowing RSS to shrink after traffic spikes subside. Together these ensure that: - Per-request memory is released promptly after logging completes - The OS reclaims freed heap memory periodically, so RSS drops during idle periods instead of staying at the peak level indefinitely Co-authored-by: Ishaan Jaff <ishaan-jaff@users.noreply.github.com>
This commit is contained in:
parent
cc85fe5921
commit
d919b1048e
4 changed files with 128 additions and 1 deletions
23
AGENTS.md
23
AGENTS.md
|
|
@ -189,4 +189,25 @@ When opening issues or pull requests, follow these templates:
|
|||
- Check similar provider implementations
|
||||
- Ensure comprehensive test coverage
|
||||
- Update documentation appropriately
|
||||
- Consider backward compatibility impact
|
||||
- Consider backward compatibility impact
|
||||
|
||||
## Cursor Cloud specific instructions
|
||||
|
||||
### Environment setup
|
||||
- Python 3.12 with `poetry install --with dev,proxy-dev --extras proxy`
|
||||
- Install `psycopg-binary` and `psutil` separately: `poetry run pip install psycopg-binary psutil`
|
||||
- Lint: `cd litellm && poetry run ruff check .`
|
||||
- Unit tests: `poetry run pytest tests/test_litellm/ -x -n 4` (see `Makefile` for specific test groups)
|
||||
- Memory/load tests: `poetry run pytest tests/load_tests/test_linear_memory_growth.py::test_memory_baseline_1k -v` (run individually, not all at once)
|
||||
|
||||
### Running the proxy server locally
|
||||
- `poetry run litellm --model openai/gpt-4o` (requires `OPENAI_API_KEY`)
|
||||
- Proxy listens on port 4000 by default
|
||||
- No PostgreSQL or Redis required for basic SDK/proxy testing with mock endpoints
|
||||
|
||||
### Memory leak testing
|
||||
- Existing tests in `tests/load_tests/` use a local mock OpenAI server (no API keys needed)
|
||||
- `test_linear_memory_growth.py` tests are designed to run **individually** (memory baselines drift when combined)
|
||||
- `_periodic_memory_cleanup()` in `litellm/proxy/common_utils/memory_utils.py` calls `gc.collect()` + `malloc_trim()` on Linux to return freed memory to the OS
|
||||
- `LITELLM_MEMORY_CLEANUP_INTERVAL` env var controls cleanup frequency (default 60s)
|
||||
- `PYTHON_GC_THRESHOLD` env var configures GC thresholds (format: `gen0,gen1,gen2`)
|
||||
|
|
@ -2368,6 +2368,9 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
str(e)
|
||||
),
|
||||
)
|
||||
finally:
|
||||
# Release large data structures now that all callbacks have consumed them
|
||||
self._cleanup_after_logging()
|
||||
|
||||
async def async_success_handler( # noqa: PLR0915
|
||||
self, result=None, start_time=None, end_time=None, cache_hit=None, **kwargs
|
||||
|
|
@ -2713,6 +2716,45 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
self._handle_callback_failure(callback=callback)
|
||||
pass
|
||||
|
||||
# Release large data structures now that all callbacks have consumed them
|
||||
self._cleanup_after_logging()
|
||||
|
||||
def _cleanup_after_logging(self):
|
||||
"""
|
||||
Release large data structures after all logging callbacks have completed.
|
||||
|
||||
This prevents memory from accumulating when many requests are processed,
|
||||
especially under high concurrency. The Logging object may still be
|
||||
referenced by async tasks, but its large payload fields are no longer
|
||||
needed after callbacks have consumed them.
|
||||
|
||||
Fields intentionally kept:
|
||||
- ``response_cost``, ``standard_logging_object`` – may be read after
|
||||
callbacks (e.g. for response headers).
|
||||
- ``async_complete_streaming_response``, ``complete_streaming_response``
|
||||
– used as guard flags to prevent double-processing on repeated calls.
|
||||
"""
|
||||
# Clear streaming chunks (can be very large for long responses)
|
||||
if hasattr(self, "streaming_chunks"):
|
||||
self.streaming_chunks.clear()
|
||||
if hasattr(self, "sync_streaming_chunks"):
|
||||
self.sync_streaming_chunks.clear()
|
||||
|
||||
# Clear large payload-only fields from model_call_details.
|
||||
# These hold the raw request input and response text which can be
|
||||
# substantial (kilobytes to megabytes per request).
|
||||
_large_keys = (
|
||||
"input",
|
||||
"original_response",
|
||||
"complete_response",
|
||||
"raw_request_typed_dict",
|
||||
)
|
||||
for key in _large_keys:
|
||||
self.model_call_details.pop(key, None)
|
||||
|
||||
# Clear messages reference (the full request prompt)
|
||||
self.messages = None
|
||||
|
||||
def _handle_callback_failure(self, callback: Any):
|
||||
"""
|
||||
Handle callback logging failures by incrementing Prometheus metrics.
|
||||
|
|
|
|||
47
litellm/proxy/common_utils/memory_utils.py
Normal file
47
litellm/proxy/common_utils/memory_utils.py
Normal file
|
|
@ -0,0 +1,47 @@
|
|||
"""
|
||||
Periodic memory cleanup utilities for the LiteLLM proxy.
|
||||
|
||||
Python's memory allocator (pymalloc) does not return freed memory arenas back
|
||||
to the OS by default. After handling a burst of traffic, process RSS stays
|
||||
elevated even though the objects have been garbage-collected. This module
|
||||
provides a lightweight scheduled job that:
|
||||
|
||||
1. Runs a full GC collection cycle
|
||||
2. Calls ``malloc_trim(0)`` on Linux (glibc) to return unused heap pages to
|
||||
the OS
|
||||
|
||||
This is especially important for long-running proxy deployments where memory
|
||||
grows during load tests / peak traffic and never shrinks back.
|
||||
|
||||
Environment variables:
|
||||
LITELLM_MEMORY_CLEANUP_INTERVAL – seconds between cleanup runs (default 60)
|
||||
"""
|
||||
|
||||
import ctypes
|
||||
import gc
|
||||
import sys
|
||||
|
||||
from litellm._logging import verbose_proxy_logger
|
||||
|
||||
_libc = None
|
||||
_malloc_trim_available = False
|
||||
|
||||
if sys.platform == "linux":
|
||||
try:
|
||||
_libc = ctypes.CDLL("libc.so.6")
|
||||
_malloc_trim_available = hasattr(_libc, "malloc_trim")
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def _periodic_memory_cleanup() -> None:
|
||||
"""Run GC and release freed memory back to the OS."""
|
||||
gc.collect()
|
||||
|
||||
if _malloc_trim_available and _libc is not None:
|
||||
try:
|
||||
_libc.malloc_trim(0)
|
||||
except Exception as exc:
|
||||
verbose_proxy_logger.debug(
|
||||
"malloc_trim failed (non-critical): %s", exc
|
||||
)
|
||||
|
|
@ -5667,6 +5667,23 @@ class ProxyStartupEvent:
|
|||
misfire_grace_time=APSCHEDULER_MISFIRE_GRACE_TIME,
|
||||
)
|
||||
|
||||
### PERIODIC MEMORY CLEANUP ###
|
||||
# Return freed memory to the OS. Python's allocator holds freed arenas
|
||||
# indefinitely; malloc_trim() on Linux releases them, preventing RSS
|
||||
# from staying elevated after traffic spikes.
|
||||
from litellm.proxy.common_utils.memory_utils import (
|
||||
_periodic_memory_cleanup,
|
||||
)
|
||||
|
||||
scheduler.add_job(
|
||||
_periodic_memory_cleanup,
|
||||
"interval",
|
||||
seconds=int(os.getenv("LITELLM_MEMORY_CLEANUP_INTERVAL", 60)),
|
||||
id="memory_cleanup_job",
|
||||
replace_existing=True,
|
||||
misfire_grace_time=APSCHEDULER_MISFIRE_GRACE_TIME,
|
||||
)
|
||||
|
||||
### MONITOR SPEND LOGS QUEUE (queue-size-based job) ###
|
||||
if general_settings.get("disable_spend_logs", False) is False:
|
||||
from litellm.proxy.utils import _monitor_spend_logs_queue
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue