fix(slack_alerting): alert once per hanging request

The min-age gate stops false positives for young in-flight requests, but
a genuinely hanging request still re-alerted on every checker tick within
the cache TTL. With the wider TTL (1.5x threshold + 60s) that is 1-2 extra
Slack notifications per stuck request at the default 600s threshold.

Flag a HangingRequestData entry as alerted once its alert fires and skip
flagged entries on later ticks, so each hang produces exactly one alert.
The cache reference is mutated in place, so the TTL is untouched and still
handles cleanup. Adds a regression test asserting one alert across multiple
ticks.

Fixes #27855.
This commit is contained in:
Filippo Mattia Menghi 2026-06-11 10:31:41 +02:00
parent 39a25b84f4
commit da7964c3de
3 changed files with 44 additions and 0 deletions

View file

@ -113,6 +113,9 @@ class AlertingHangingRequestCheck:
if hanging_request_data is None:
continue
if hanging_request_data.alerted:
continue
request_status = (
await proxy_logging_obj.internal_usage_cache.async_get_cache(
key="request_status:{}".format(hanging_request_data.request_id),
@ -141,6 +144,9 @@ class AlertingHangingRequestCheck:
await self.send_hanging_request_alert(
hanging_request_data=hanging_request_data
)
# flag so the entry is skipped on later ticks; one alert per hang,
# with the existing TTL still handling cleanup
hanging_request_data.alerted = True
return

View file

@ -203,6 +203,7 @@ class HangingRequestData(BaseModel):
team_alias: Optional[str] = None
alerting_metadata: Optional[dict] = None
created_at: float = Field(default_factory=time.time)
alerted: bool = False
class AlertTypeConfig(LiteLLMPydanticObjectBase):

View file

@ -238,6 +238,43 @@ class TestAlertingHangingRequestCheck:
# Verify alert was sent for hanging request
hanging_request_checker.slack_alerting_object.send_alert.assert_called_once()
@pytest.mark.asyncio
async def test_send_alerts_for_hanging_requests_alerts_once_per_hang(
self, hanging_request_checker
):
"""
A single hanging request must alert exactly once even though the
checker tick revisits it on every run within the cache TTL.
"""
hanging_data = HangingRequestData(
request_id="hanging_once_555",
model="gpt-4",
api_base="https://api.openai.com/v1",
created_at=time.time() - 301,
)
await hanging_request_checker.hanging_request_cache.async_set_cache(
key="hanging_once_555", value=hanging_data, ttl=300
)
with patch("litellm.proxy.proxy_server.proxy_logging_obj") as mock_proxy:
mock_internal_cache = AsyncMock()
mock_internal_cache.async_get_cache.return_value = None
mock_proxy.internal_usage_cache = mock_internal_cache
hanging_request_checker.hanging_request_cache.async_get_oldest_n_keys = (
AsyncMock(return_value=["hanging_once_555"])
)
for _ in range(3):
await hanging_request_checker.send_alerts_for_hanging_requests()
assert hanging_request_checker.slack_alerting_object.send_alert.call_count == 1
cached = await hanging_request_checker.hanging_request_cache.async_get_cache(
key="hanging_once_555"
)
assert cached is not None
assert cached.alerted is True
@pytest.mark.asyncio
async def test_send_alerts_for_hanging_requests_skips_request_younger_than_threshold(
self, hanging_request_checker