From 653aff65271a463c41b11af3c6bf1cdedba55cfa Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Tue, 16 Jul 2024 08:53:46 -0700 Subject: [PATCH 1/3] fix storing request status in mem --- litellm/proxy/utils.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/litellm/proxy/utils.py b/litellm/proxy/utils.py index a7607fbadd9..4669381976d 100644 --- a/litellm/proxy/utils.py +++ b/litellm/proxy/utils.py @@ -284,11 +284,14 @@ class ProxyLogging: if self.alerting is None: return + # current alerting threshold + alerting_threshold = self.alerting_threshold or 120 + await self.internal_usage_cache.async_set_cache( key="request_status:{}".format(litellm_call_id), value=status, local_only=True, - ttl=120, + ttl=alerting_threshold, ) # The actual implementation of the function From 769843845de2f73ecdc5f4f4969144f1ef5b6f5f Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Tue, 16 Jul 2024 17:12:59 -0700 Subject: [PATCH 2/3] fix tracking hanging requests --- litellm/proxy/utils.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/litellm/proxy/utils.py b/litellm/proxy/utils.py index 4669381976d..73a15886e0b 100644 --- a/litellm/proxy/utils.py +++ b/litellm/proxy/utils.py @@ -285,7 +285,8 @@ class ProxyLogging: return # current alerting threshold - alerting_threshold = self.alerting_threshold or 120 + # add a 100 second buffer to the alerting threshold + alerting_threshold = self.alerting_threshold or 120 + 100 await self.internal_usage_cache.async_set_cache( key="request_status:{}".format(litellm_call_id), From f9f2f3baa0b549e6410eba81602e021e8e6b1721 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Tue, 16 Jul 2024 20:31:56 -0700 Subject: [PATCH 3/3] fix calculate correct alerting threshold --- litellm/proxy/utils.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/litellm/proxy/utils.py b/litellm/proxy/utils.py index 73a15886e0b..2145d1226d3 100644 --- a/litellm/proxy/utils.py +++ b/litellm/proxy/utils.py @@ -285,8 +285,11 @@ class ProxyLogging: return # current alerting threshold + alerting_threshold: float = self.alerting_threshold + # add a 100 second buffer to the alerting threshold - alerting_threshold = self.alerting_threshold or 120 + 100 + # ensures we don't send errant hanging request slack alerts + alerting_threshold += 100 await self.internal_usage_cache.async_set_cache( key="request_status:{}".format(litellm_call_id),