diff --git a/litellm/integrations/SlackAlerting/hanging_request_check.py b/litellm/integrations/SlackAlerting/hanging_request_check.py index 4d7cbfe8fd1..85670610242 100644 --- a/litellm/integrations/SlackAlerting/hanging_request_check.py +++ b/litellm/integrations/SlackAlerting/hanging_request_check.py @@ -138,6 +138,13 @@ class AlertingHangingRequestCheck: # flag so the entry is skipped on later ticks; one alert per hang, # with the existing TTL still handling cleanup hanging_request_data.alerted = True + # InMemoryCache returns read-isolated mutable values. Persist the + # state transition explicitly so the next checker tick observes + # that this request has already been alerted. + await self.hanging_request_cache.async_set_cache( + key=request_id, + value=hanging_request_data, + ) return