mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-12 23:01:41 +00:00
fix(logging): release httpx.Response/Headers after callbacks complete
The Logging object's model_call_details retains full httpx.Response objects (key: 'httpx_response') and httpx.Headers (key: 'response_headers') for every LLM request. These contain OrderedDict instances that accumulate when the Logging object isn't promptly garbage-collected (e.g. referenced by pending asyncio tasks). Added _cleanup_heavy_references() that runs at the end of all four logging paths (success_handler, async_success_handler, failure_handler, async_failure_handler) to explicitly pop heavyweight references from model_call_details. This is the primary fix for the OrderedDict leak observed growing from 18K to 155K objects. Co-authored-by: Ishaan Jaff <ishaan-jaff@users.noreply.github.com>
This commit is contained in:
parent
c977230bc2
commit
b7d41d49bd
1 changed files with 23 additions and 0 deletions
|
|
@ -2306,6 +2306,8 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
str(e)
|
||||
),
|
||||
)
|
||||
finally:
|
||||
self._cleanup_heavy_references()
|
||||
|
||||
async def async_success_handler( # noqa: PLR0915
|
||||
self, result=None, start_time=None, end_time=None, cache_hit=None, **kwargs
|
||||
|
|
@ -2626,6 +2628,23 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
self._handle_callback_failure(callback=callback)
|
||||
pass
|
||||
|
||||
self._cleanup_heavy_references()
|
||||
|
||||
def _cleanup_heavy_references(self) -> None:
|
||||
"""Release heavyweight objects from model_call_details after logging.
|
||||
|
||||
httpx.Response objects and their Headers hold OrderedDicts that
|
||||
accumulate if the Logging object isn't promptly garbage-collected
|
||||
(e.g. when referenced by pending asyncio tasks).
|
||||
"""
|
||||
for key in (
|
||||
"httpx_response",
|
||||
"response_headers",
|
||||
"raw_request_typed_dict",
|
||||
"complete_streaming_response",
|
||||
):
|
||||
self.model_call_details.pop(key, None)
|
||||
|
||||
def _handle_callback_failure(self, callback: Any):
|
||||
"""
|
||||
Handle callback logging failures by incrementing Prometheus metrics.
|
||||
|
|
@ -2921,6 +2940,8 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
str(e)
|
||||
)
|
||||
)
|
||||
finally:
|
||||
self._cleanup_heavy_references()
|
||||
|
||||
async def async_failure_handler(
|
||||
self, exception, traceback_exception, start_time=None, end_time=None
|
||||
|
|
@ -2986,6 +3007,8 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
# Track callback logging failures in Prometheus
|
||||
self._handle_callback_failure(callback=callback)
|
||||
|
||||
self._cleanup_heavy_references()
|
||||
|
||||
def _get_trace_id(self, service_name: Literal["langfuse"]) -> Optional[str]:
|
||||
"""
|
||||
For the given service (e.g. langfuse), return the trace_id actually logged.
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue