mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
fix(slack_alerting): alert once per hanging request
The min-age gate stops false positives for young in-flight requests, but a genuinely hanging request still re-alerted on every checker tick within the cache TTL. With the wider TTL (1.5x threshold + 60s) that is 1-2 extra Slack notifications per stuck request at the default 600s threshold. Flag a HangingRequestData entry as alerted once its alert fires and skip flagged entries on later ticks, so each hang produces exactly one alert. The cache reference is mutated in place, so the TTL is untouched and still handles cleanup. Adds a regression test asserting one alert across multiple ticks. Fixes #27855.
This commit is contained in:
parent
39a25b84f4
commit
da7964c3de
3 changed files with 44 additions and 0 deletions
|
|
@ -113,6 +113,9 @@ class AlertingHangingRequestCheck:
|
|||
if hanging_request_data is None:
|
||||
continue
|
||||
|
||||
if hanging_request_data.alerted:
|
||||
continue
|
||||
|
||||
request_status = (
|
||||
await proxy_logging_obj.internal_usage_cache.async_get_cache(
|
||||
key="request_status:{}".format(hanging_request_data.request_id),
|
||||
|
|
@ -141,6 +144,9 @@ class AlertingHangingRequestCheck:
|
|||
await self.send_hanging_request_alert(
|
||||
hanging_request_data=hanging_request_data
|
||||
)
|
||||
# flag so the entry is skipped on later ticks; one alert per hang,
|
||||
# with the existing TTL still handling cleanup
|
||||
hanging_request_data.alerted = True
|
||||
|
||||
return
|
||||
|
||||
|
|
|
|||
|
|
@ -203,6 +203,7 @@ class HangingRequestData(BaseModel):
|
|||
team_alias: Optional[str] = None
|
||||
alerting_metadata: Optional[dict] = None
|
||||
created_at: float = Field(default_factory=time.time)
|
||||
alerted: bool = False
|
||||
|
||||
|
||||
class AlertTypeConfig(LiteLLMPydanticObjectBase):
|
||||
|
|
|
|||
|
|
@ -238,6 +238,43 @@ class TestAlertingHangingRequestCheck:
|
|||
# Verify alert was sent for hanging request
|
||||
hanging_request_checker.slack_alerting_object.send_alert.assert_called_once()
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_send_alerts_for_hanging_requests_alerts_once_per_hang(
|
||||
self, hanging_request_checker
|
||||
):
|
||||
"""
|
||||
A single hanging request must alert exactly once even though the
|
||||
checker tick revisits it on every run within the cache TTL.
|
||||
"""
|
||||
hanging_data = HangingRequestData(
|
||||
request_id="hanging_once_555",
|
||||
model="gpt-4",
|
||||
api_base="https://api.openai.com/v1",
|
||||
created_at=time.time() - 301,
|
||||
)
|
||||
await hanging_request_checker.hanging_request_cache.async_set_cache(
|
||||
key="hanging_once_555", value=hanging_data, ttl=300
|
||||
)
|
||||
|
||||
with patch("litellm.proxy.proxy_server.proxy_logging_obj") as mock_proxy:
|
||||
mock_internal_cache = AsyncMock()
|
||||
mock_internal_cache.async_get_cache.return_value = None
|
||||
mock_proxy.internal_usage_cache = mock_internal_cache
|
||||
|
||||
hanging_request_checker.hanging_request_cache.async_get_oldest_n_keys = (
|
||||
AsyncMock(return_value=["hanging_once_555"])
|
||||
)
|
||||
|
||||
for _ in range(3):
|
||||
await hanging_request_checker.send_alerts_for_hanging_requests()
|
||||
|
||||
assert hanging_request_checker.slack_alerting_object.send_alert.call_count == 1
|
||||
cached = await hanging_request_checker.hanging_request_cache.async_get_cache(
|
||||
key="hanging_once_555"
|
||||
)
|
||||
assert cached is not None
|
||||
assert cached.alerted is True
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_send_alerts_for_hanging_requests_skips_request_younger_than_threshold(
|
||||
self, hanging_request_checker
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue