mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
fix(proxy): completion-marker ttl outliving hanging-request tracker entry
update_request_status wrote the "request completed" marker with ttl = alerting_threshold + 100, while the hanging-request tracker entry (HangingRequestCheck) lives for alerting_threshold * 1.5 + HANGING_ALERT_BUFFER_TIME_SECONDS. Since the check loop only scans the 20 oldest tracked entries per pass, a sustained burst above ~8 req/min can let a completed request's tracker entry sit in the backlog long enough for its shorter-lived completion marker to expire first - the check then finds no marker, assumes the request is still running, and fires a false "hanging request" Slack alert for a request that already finished in under a second. Align the marker's ttl with the tracker entry's formula so a completed request's marker always outlives the window in which it might still be checked. Fixes #43285
This commit is contained in:
parent
dd63637322
commit
b90e9c95af
2 changed files with 41 additions and 3 deletions
|
|
@ -1434,9 +1434,18 @@ class ProxyLogging:
|
|||
# current alerting threshold
|
||||
alerting_threshold: float = self.alerting_threshold
|
||||
|
||||
# add a 100 second buffer to the alerting threshold
|
||||
# ensures we don't send errant hanging request slack alerts
|
||||
alerting_threshold += 100
|
||||
# This marker's ttl must outlive the hanging-request tracker entry
|
||||
# (alerting_threshold * 1.5 + HANGING_ALERT_BUFFER_TIME_SECONDS, see
|
||||
# HangingRequestCheck._get_metadata) - otherwise a completed request
|
||||
# can have its marker expire before the tracker gets around to
|
||||
# checking it (only the 20 oldest entries are scanned per pass),
|
||||
# producing a false "hanging request" alert for a request that
|
||||
# already finished successfully.
|
||||
from litellm.types.integrations.slack_alerting import (
|
||||
HANGING_ALERT_BUFFER_TIME_SECONDS,
|
||||
)
|
||||
|
||||
alerting_threshold = alerting_threshold * 1.5 + HANGING_ALERT_BUFFER_TIME_SECONDS
|
||||
|
||||
await self.internal_usage_cache.async_set_cache(
|
||||
key=f"request_status:{litellm_call_id}",
|
||||
|
|
|
|||
|
|
@ -26,6 +26,35 @@ def test_get_custom_url(monkeypatch):
|
|||
assert custom_url == "http://0.0.0.0:4000/litellm/ui/"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_update_request_status_marker_ttl_outlives_tracker_entry():
|
||||
"""The completion marker written by update_request_status must have a
|
||||
ttl >= the hanging-request tracker entry's ttl (alerting_threshold * 1.5
|
||||
+ HANGING_ALERT_BUFFER_TIME_SECONDS). Otherwise a completed request's
|
||||
marker can expire before the hanging-request check gets around to
|
||||
scanning it, producing a false "hanging request" alert. See #43285.
|
||||
"""
|
||||
from litellm.types.integrations.slack_alerting import (
|
||||
HANGING_ALERT_BUFFER_TIME_SECONDS,
|
||||
)
|
||||
|
||||
proxy_logging_obj = ProxyLogging(user_api_key_cache=DualCache())
|
||||
proxy_logging_obj.alerting = ["slack"]
|
||||
proxy_logging_obj.alerting_threshold = 300
|
||||
|
||||
tracker_entry_ttl = 300 * 1.5 + HANGING_ALERT_BUFFER_TIME_SECONDS
|
||||
|
||||
with patch.object(
|
||||
proxy_logging_obj.internal_usage_cache, "async_set_cache", new=AsyncMock()
|
||||
) as mock_set_cache:
|
||||
await proxy_logging_obj.update_request_status(
|
||||
litellm_call_id="test_call_id", status="success"
|
||||
)
|
||||
|
||||
used_ttl = mock_set_cache.call_args.kwargs["ttl"]
|
||||
assert used_ttl >= tracker_entry_ttl
|
||||
|
||||
|
||||
def test_proxy_only_error_true_for_llm_route():
|
||||
proxy_logging_obj = ProxyLogging(user_api_key_cache=DualCache())
|
||||
assert proxy_logging_obj._is_proxy_only_llm_api_error(
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue