fix(slack_alerting): skip hanging request alerts below the threshold

The hanging request check alerted on any cached request whose
completion status was not yet recorded, with no minimum age check.
Since the background loop runs every alerting_threshold / 2 seconds,
any request that happened to be in flight at a check fired a
"hanging - Ns+ request time" alert even if it was only seconds old,
producing a steady stream of false positives.

Add a created_at timestamp to HangingRequestData, stamped when the
request enters the hanging request cache, and skip requests younger
than alerting_threshold without evicting them, so a later check can
still alert if they never complete. Extend the cache TTL from
threshold + 60s to 1.5x threshold + 60s; with the age check, entries
only become alertable after threshold seconds, and the check period
is threshold / 2, so the old TTL could evict a genuinely hanging
request before any check saw it cross the threshold.

Fixes #27855.
This commit is contained in:
Filippo Mattia Menghi 2026-06-10 10:21:36 +02:00
parent e15b37a18e
commit 39a25b84f4
3 changed files with 64 additions and 13 deletions

View file

@ -8,6 +8,7 @@ Notes:
"""
import asyncio
import time
from typing import TYPE_CHECKING, Any, Optional
import litellm
@ -36,11 +37,15 @@ class AlertingHangingRequestCheck:
slack_alerting_object: SlackAlerting,
):
self.slack_alerting_object = slack_alerting_object
# checks run every alerting_threshold / 2 seconds, so entries must
# stay cached for at least 1.5x the threshold to guarantee a check
# happens after they cross it
self.hanging_request_cache_ttl = int(
self.slack_alerting_object.alerting_threshold * 1.5
+ HANGING_ALERT_BUFFER_TIME_SECONDS
)
self.hanging_request_cache = InMemoryCache(
default_ttl=int(
self.slack_alerting_object.alerting_threshold
+ HANGING_ALERT_BUFFER_TIME_SECONDS
),
default_ttl=self.hanging_request_cache_ttl,
)
async def add_request_to_hanging_request_check(
@ -76,10 +81,7 @@ class AlertingHangingRequestCheck:
await self.hanging_request_cache.async_set_cache(
key=hanging_request_data.request_id,
value=hanging_request_data,
ttl=int(
self.slack_alerting_object.alerting_threshold
+ HANGING_ALERT_BUFFER_TIME_SECONDS
),
ttl=self.hanging_request_cache_ttl,
)
return
@ -127,6 +129,12 @@ class AlertingHangingRequestCheck:
)
continue
request_age_seconds = time.time() - hanging_request_data.created_at
if request_age_seconds < self.slack_alerting_object.alerting_threshold:
# in-flight but below the alerting threshold; keep it cached
# so a later check can alert if it never completes
continue
################
# Send the Alert on Slack
################

View file

@ -1,4 +1,5 @@
import os
import time
from datetime import datetime as dt
from enum import Enum
from typing import Any, Dict, List, Literal, Optional, Set, Union
@ -201,6 +202,7 @@ class HangingRequestData(BaseModel):
key_alias: Optional[str] = None
team_alias: Optional[str] = None
alerting_metadata: Optional[dict] = None
created_at: float = Field(default_factory=time.time)
class AlertTypeConfig(LiteLLMPydanticObjectBase):

View file

@ -1,6 +1,7 @@
import json
import os
import sys
import time
from typing import Optional
from unittest.mock import AsyncMock, MagicMock, patch
@ -35,13 +36,13 @@ class TestAlertingHangingRequestCheck:
async def test_init_creates_cache_with_correct_ttl(self, mock_slack_alerting):
"""
Test that initialization creates a hanging request cache with correct TTL.
The TTL should be alerting_threshold + buffer time.
The TTL should be 1.5x alerting_threshold + buffer time, so entries
survive long enough to be checked after crossing the threshold.
"""
checker = AlertingHangingRequestCheck(slack_alerting_object=mock_slack_alerting)
# The cache should be created with TTL = alerting_threshold + buffer time
expected_ttl = (
mock_slack_alerting.alerting_threshold + 60
expected_ttl = int(
mock_slack_alerting.alerting_threshold * 1.5 + 60
) # HANGING_ALERT_BUFFER_TIME_SECONDS
assert checker.hanging_request_cache.default_ttl == expected_ttl
@ -208,13 +209,14 @@ class TestAlertingHangingRequestCheck:
Test send_alerts_for_hanging_requests when request is actually hanging.
Should send alert for requests that haven't completed within threshold.
"""
# Add a hanging request to the cache
# Add a hanging request that is older than the alerting threshold
hanging_data = HangingRequestData(
request_id="hanging_request_999",
model="gpt-4",
api_base="https://api.openai.com/v1",
key_alias="test_key",
team_alias="test_team",
created_at=time.time() - 301,
)
await hanging_request_checker.hanging_request_cache.async_set_cache(
key="hanging_request_999", value=hanging_data, ttl=300
@ -236,6 +238,45 @@ class TestAlertingHangingRequestCheck:
# Verify alert was sent for hanging request
hanging_request_checker.slack_alerting_object.send_alert.assert_called_once()
@pytest.mark.asyncio
async def test_send_alerts_for_hanging_requests_skips_request_younger_than_threshold(
self, hanging_request_checker
):
"""
Test that an in-flight request younger than the alerting threshold
does not trigger an alert and stays in the cache for later checks.
"""
hanging_data = HangingRequestData(
request_id="young_request_123",
model="gpt-4",
api_base="https://api.openai.com/v1",
)
await hanging_request_checker.hanging_request_cache.async_set_cache(
key="young_request_123", value=hanging_data, ttl=300
)
with patch("litellm.proxy.proxy_server.proxy_logging_obj") as mock_proxy:
# Mock internal usage cache to return None (request still in flight)
mock_internal_cache = AsyncMock()
mock_internal_cache.async_get_cache.return_value = None
mock_proxy.internal_usage_cache = mock_internal_cache
hanging_request_checker.hanging_request_cache.async_get_oldest_n_keys = (
AsyncMock(return_value=["young_request_123"])
)
await hanging_request_checker.send_alerts_for_hanging_requests()
# No alert for a request below the threshold, and it must remain
# cached so a later check can alert if it never completes
hanging_request_checker.slack_alerting_object.send_alert.assert_not_called()
assert (
await hanging_request_checker.hanging_request_cache.async_get_cache(
key="young_request_123"
)
is not None
)
@pytest.mark.asyncio
async def test_send_alerts_for_hanging_requests_with_missing_hanging_data(
self, hanging_request_checker