mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-05 02:41:56 +00:00
fix(slack_alerting): skip hanging request alerts below the threshold
The hanging request check alerted on any cached request whose completion status was not yet recorded, with no minimum age check. Since the background loop runs every alerting_threshold / 2 seconds, any request that happened to be in flight at a check fired a "hanging - Ns+ request time" alert even if it was only seconds old, producing a steady stream of false positives. Add a created_at timestamp to HangingRequestData, stamped when the request enters the hanging request cache, and skip requests younger than alerting_threshold without evicting them, so a later check can still alert if they never complete. Extend the cache TTL from threshold + 60s to 1.5x threshold + 60s; with the age check, entries only become alertable after threshold seconds, and the check period is threshold / 2, so the old TTL could evict a genuinely hanging request before any check saw it cross the threshold. Fixes #27855.
This commit is contained in:
parent
e15b37a18e
commit
39a25b84f4
3 changed files with 64 additions and 13 deletions
|
|
@ -8,6 +8,7 @@ Notes:
|
|||
"""
|
||||
|
||||
import asyncio
|
||||
import time
|
||||
from typing import TYPE_CHECKING, Any, Optional
|
||||
|
||||
import litellm
|
||||
|
|
@ -36,11 +37,15 @@ class AlertingHangingRequestCheck:
|
|||
slack_alerting_object: SlackAlerting,
|
||||
):
|
||||
self.slack_alerting_object = slack_alerting_object
|
||||
# checks run every alerting_threshold / 2 seconds, so entries must
|
||||
# stay cached for at least 1.5x the threshold to guarantee a check
|
||||
# happens after they cross it
|
||||
self.hanging_request_cache_ttl = int(
|
||||
self.slack_alerting_object.alerting_threshold * 1.5
|
||||
+ HANGING_ALERT_BUFFER_TIME_SECONDS
|
||||
)
|
||||
self.hanging_request_cache = InMemoryCache(
|
||||
default_ttl=int(
|
||||
self.slack_alerting_object.alerting_threshold
|
||||
+ HANGING_ALERT_BUFFER_TIME_SECONDS
|
||||
),
|
||||
default_ttl=self.hanging_request_cache_ttl,
|
||||
)
|
||||
|
||||
async def add_request_to_hanging_request_check(
|
||||
|
|
@ -76,10 +81,7 @@ class AlertingHangingRequestCheck:
|
|||
await self.hanging_request_cache.async_set_cache(
|
||||
key=hanging_request_data.request_id,
|
||||
value=hanging_request_data,
|
||||
ttl=int(
|
||||
self.slack_alerting_object.alerting_threshold
|
||||
+ HANGING_ALERT_BUFFER_TIME_SECONDS
|
||||
),
|
||||
ttl=self.hanging_request_cache_ttl,
|
||||
)
|
||||
return
|
||||
|
||||
|
|
@ -127,6 +129,12 @@ class AlertingHangingRequestCheck:
|
|||
)
|
||||
continue
|
||||
|
||||
request_age_seconds = time.time() - hanging_request_data.created_at
|
||||
if request_age_seconds < self.slack_alerting_object.alerting_threshold:
|
||||
# in-flight but below the alerting threshold; keep it cached
|
||||
# so a later check can alert if it never completes
|
||||
continue
|
||||
|
||||
################
|
||||
# Send the Alert on Slack
|
||||
################
|
||||
|
|
|
|||
|
|
@ -1,4 +1,5 @@
|
|||
import os
|
||||
import time
|
||||
from datetime import datetime as dt
|
||||
from enum import Enum
|
||||
from typing import Any, Dict, List, Literal, Optional, Set, Union
|
||||
|
|
@ -201,6 +202,7 @@ class HangingRequestData(BaseModel):
|
|||
key_alias: Optional[str] = None
|
||||
team_alias: Optional[str] = None
|
||||
alerting_metadata: Optional[dict] = None
|
||||
created_at: float = Field(default_factory=time.time)
|
||||
|
||||
|
||||
class AlertTypeConfig(LiteLLMPydanticObjectBase):
|
||||
|
|
|
|||
|
|
@ -1,6 +1,7 @@
|
|||
import json
|
||||
import os
|
||||
import sys
|
||||
import time
|
||||
from typing import Optional
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
|
|
@ -35,13 +36,13 @@ class TestAlertingHangingRequestCheck:
|
|||
async def test_init_creates_cache_with_correct_ttl(self, mock_slack_alerting):
|
||||
"""
|
||||
Test that initialization creates a hanging request cache with correct TTL.
|
||||
The TTL should be alerting_threshold + buffer time.
|
||||
The TTL should be 1.5x alerting_threshold + buffer time, so entries
|
||||
survive long enough to be checked after crossing the threshold.
|
||||
"""
|
||||
checker = AlertingHangingRequestCheck(slack_alerting_object=mock_slack_alerting)
|
||||
|
||||
# The cache should be created with TTL = alerting_threshold + buffer time
|
||||
expected_ttl = (
|
||||
mock_slack_alerting.alerting_threshold + 60
|
||||
expected_ttl = int(
|
||||
mock_slack_alerting.alerting_threshold * 1.5 + 60
|
||||
) # HANGING_ALERT_BUFFER_TIME_SECONDS
|
||||
assert checker.hanging_request_cache.default_ttl == expected_ttl
|
||||
|
||||
|
|
@ -208,13 +209,14 @@ class TestAlertingHangingRequestCheck:
|
|||
Test send_alerts_for_hanging_requests when request is actually hanging.
|
||||
Should send alert for requests that haven't completed within threshold.
|
||||
"""
|
||||
# Add a hanging request to the cache
|
||||
# Add a hanging request that is older than the alerting threshold
|
||||
hanging_data = HangingRequestData(
|
||||
request_id="hanging_request_999",
|
||||
model="gpt-4",
|
||||
api_base="https://api.openai.com/v1",
|
||||
key_alias="test_key",
|
||||
team_alias="test_team",
|
||||
created_at=time.time() - 301,
|
||||
)
|
||||
await hanging_request_checker.hanging_request_cache.async_set_cache(
|
||||
key="hanging_request_999", value=hanging_data, ttl=300
|
||||
|
|
@ -236,6 +238,45 @@ class TestAlertingHangingRequestCheck:
|
|||
# Verify alert was sent for hanging request
|
||||
hanging_request_checker.slack_alerting_object.send_alert.assert_called_once()
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_send_alerts_for_hanging_requests_skips_request_younger_than_threshold(
|
||||
self, hanging_request_checker
|
||||
):
|
||||
"""
|
||||
Test that an in-flight request younger than the alerting threshold
|
||||
does not trigger an alert and stays in the cache for later checks.
|
||||
"""
|
||||
hanging_data = HangingRequestData(
|
||||
request_id="young_request_123",
|
||||
model="gpt-4",
|
||||
api_base="https://api.openai.com/v1",
|
||||
)
|
||||
await hanging_request_checker.hanging_request_cache.async_set_cache(
|
||||
key="young_request_123", value=hanging_data, ttl=300
|
||||
)
|
||||
|
||||
with patch("litellm.proxy.proxy_server.proxy_logging_obj") as mock_proxy:
|
||||
# Mock internal usage cache to return None (request still in flight)
|
||||
mock_internal_cache = AsyncMock()
|
||||
mock_internal_cache.async_get_cache.return_value = None
|
||||
mock_proxy.internal_usage_cache = mock_internal_cache
|
||||
|
||||
hanging_request_checker.hanging_request_cache.async_get_oldest_n_keys = (
|
||||
AsyncMock(return_value=["young_request_123"])
|
||||
)
|
||||
|
||||
await hanging_request_checker.send_alerts_for_hanging_requests()
|
||||
|
||||
# No alert for a request below the threshold, and it must remain
|
||||
# cached so a later check can alert if it never completes
|
||||
hanging_request_checker.slack_alerting_object.send_alert.assert_not_called()
|
||||
assert (
|
||||
await hanging_request_checker.hanging_request_cache.async_get_cache(
|
||||
key="young_request_123"
|
||||
)
|
||||
is not None
|
||||
)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_send_alerts_for_hanging_requests_with_missing_hanging_data(
|
||||
self, hanging_request_checker
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue