fix(proxy): completion-marker ttl outliving hanging-request tracker entry

update_request_status wrote the "request completed" marker with
ttl = alerting_threshold + 100, while the hanging-request tracker entry
(HangingRequestCheck) lives for alerting_threshold * 1.5 +
HANGING_ALERT_BUFFER_TIME_SECONDS. Since the check loop only scans the
20 oldest tracked entries per pass, a sustained burst above ~8 req/min
can let a completed request's tracker entry sit in the backlog long
enough for its shorter-lived completion marker to expire first - the
check then finds no marker, assumes the request is still running, and
fires a false "hanging request" Slack alert for a request that already
finished in under a second.

Align the marker's ttl with the tracker entry's formula so a completed
request's marker always outlives the window in which it might still be
checked.

Fixes #43285
This commit is contained in:
Neev Modh 2026-09-26 14:03:59 +05:30
parent dd63637322
commit b90e9c95af
2 changed files with 41 additions and 3 deletions

View file

@ -1434,9 +1434,18 @@ class ProxyLogging:
# current alerting threshold
alerting_threshold: float = self.alerting_threshold
# add a 100 second buffer to the alerting threshold
# ensures we don't send errant hanging request slack alerts
alerting_threshold += 100
# This marker's ttl must outlive the hanging-request tracker entry
# (alerting_threshold * 1.5 + HANGING_ALERT_BUFFER_TIME_SECONDS, see
# HangingRequestCheck._get_metadata) - otherwise a completed request
# can have its marker expire before the tracker gets around to
# checking it (only the 20 oldest entries are scanned per pass),
# producing a false "hanging request" alert for a request that
# already finished successfully.
from litellm.types.integrations.slack_alerting import (
HANGING_ALERT_BUFFER_TIME_SECONDS,
)
alerting_threshold = alerting_threshold * 1.5 + HANGING_ALERT_BUFFER_TIME_SECONDS
await self.internal_usage_cache.async_set_cache(
key=f"request_status:{litellm_call_id}",

View file

@ -26,6 +26,35 @@ def test_get_custom_url(monkeypatch):
assert custom_url == "http://0.0.0.0:4000/litellm/ui/"
@pytest.mark.asyncio
async def test_update_request_status_marker_ttl_outlives_tracker_entry():
"""The completion marker written by update_request_status must have a
ttl >= the hanging-request tracker entry's ttl (alerting_threshold * 1.5
+ HANGING_ALERT_BUFFER_TIME_SECONDS). Otherwise a completed request's
marker can expire before the hanging-request check gets around to
scanning it, producing a false "hanging request" alert. See #43285.
"""
from litellm.types.integrations.slack_alerting import (
HANGING_ALERT_BUFFER_TIME_SECONDS,
)
proxy_logging_obj = ProxyLogging(user_api_key_cache=DualCache())
proxy_logging_obj.alerting = ["slack"]
proxy_logging_obj.alerting_threshold = 300
tracker_entry_ttl = 300 * 1.5 + HANGING_ALERT_BUFFER_TIME_SECONDS
with patch.object(
proxy_logging_obj.internal_usage_cache, "async_set_cache", new=AsyncMock()
) as mock_set_cache:
await proxy_logging_obj.update_request_status(
litellm_call_id="test_call_id", status="success"
)
used_ttl = mock_set_cache.call_args.kwargs["ttl"]
assert used_ttl >= tracker_entry_ttl
def test_proxy_only_error_true_for_llm_route():
proxy_logging_obj = ProxyLogging(user_api_key_cache=DualCache())
assert proxy_logging_obj._is_proxy_only_llm_api_error(