mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
fix(slack_alerting): poll for the router inside the loop instead of a capped pre-wait
A capped pre-wait still burns the first daily pass when the router takes longer than the cap to appear (a >10 minute boot), and reads the router in two places. Folding the poll into the loop makes the first alert unconditional on boot duration and keeps a single read per pass.
This commit is contained in:
parent
6276eabf19
commit
2278118493
3 changed files with 14 additions and 13 deletions
|
|
@ -43,7 +43,6 @@ from litellm.repositories.user_repository import UserRepository
|
|||
from litellm.types.integrations.slack_alerting import *
|
||||
from litellm.types.proxy.model_deprecation import (
|
||||
DEFAULT_DEPRECATION_CHECK_INTERVAL_SECONDS,
|
||||
DEPRECATION_ROUTER_WAIT_ATTEMPTS,
|
||||
DEPRECATION_ROUTER_WAIT_SECONDS,
|
||||
)
|
||||
|
||||
|
|
@ -1083,14 +1082,12 @@ Model Info:
|
|||
self, get_llm_router: Callable[[], Router | None] = _proxy_llm_router
|
||||
) -> None:
|
||||
"""Alert once the router is loaded, then daily, re-reading the router and alert types each pass"""
|
||||
for _ in range(DEPRECATION_ROUTER_WAIT_ATTEMPTS):
|
||||
if get_llm_router() is not None:
|
||||
break
|
||||
await asyncio.sleep(DEPRECATION_ROUTER_WAIT_SECONDS)
|
||||
|
||||
while True:
|
||||
if (llm_router := get_llm_router()) is None:
|
||||
await asyncio.sleep(DEPRECATION_ROUTER_WAIT_SECONDS)
|
||||
continue
|
||||
try:
|
||||
await self.send_model_deprecation_alert(llm_router=get_llm_router())
|
||||
await self.send_model_deprecation_alert(llm_router=llm_router)
|
||||
except Exception as e: # noqa: BLE001 # a failed alert must not kill the daily loop
|
||||
verbose_proxy_logger.exception("Error in model deprecation alert loop: %s", e)
|
||||
await asyncio.sleep(DEFAULT_DEPRECATION_CHECK_INTERVAL_SECONDS)
|
||||
|
|
|
|||
|
|
@ -11,8 +11,6 @@ DEFAULT_DEPRECATION_CHECK_INTERVAL_SECONDS: Final = 24 * 60 * 60
|
|||
|
||||
DEPRECATION_ROUTER_WAIT_SECONDS: Final = 30
|
||||
|
||||
DEPRECATION_ROUTER_WAIT_ATTEMPTS: Final = 20
|
||||
|
||||
DeprecationStatus = Literal["upcoming", "imminent", "deprecated"]
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -13,7 +13,10 @@ sys.path.insert(0, os.path.abspath("../../../.."))
|
|||
import litellm
|
||||
from litellm.integrations.SlackAlerting.slack_alerting import SlackAlerting
|
||||
from litellm.proxy._types import AlertType
|
||||
from litellm.types.proxy.model_deprecation import DEPRECATION_ROUTER_WAIT_SECONDS
|
||||
from litellm.types.proxy.model_deprecation import (
|
||||
DEFAULT_DEPRECATION_CHECK_INTERVAL_SECONDS,
|
||||
DEPRECATION_ROUTER_WAIT_SECONDS,
|
||||
)
|
||||
|
||||
|
||||
def _make_router(deployments):
|
||||
|
|
@ -166,12 +169,13 @@ async def test_should_wait_for_the_router_instead_of_sleeping_a_full_day(monkeyp
|
|||
}
|
||||
]
|
||||
)
|
||||
routers = chain((None, None), repeat(router))
|
||||
router_absent_passes = 100
|
||||
routers = chain(repeat(None, router_absent_passes), repeat(router))
|
||||
slept: list[float] = []
|
||||
|
||||
async def record_sleep(seconds):
|
||||
slept.append(seconds)
|
||||
if len(slept) > 2:
|
||||
if len(slept) > router_absent_passes:
|
||||
raise asyncio.CancelledError
|
||||
|
||||
with (
|
||||
|
|
@ -186,6 +190,8 @@ async def test_should_wait_for_the_router_instead_of_sleeping_a_full_day(monkeyp
|
|||
get_llm_router=lambda: next(routers)
|
||||
)
|
||||
|
||||
assert slept[:2] == [DEPRECATION_ROUTER_WAIT_SECONDS] * 2
|
||||
assert slept == [DEPRECATION_ROUTER_WAIT_SECONDS] * router_absent_passes + [
|
||||
DEFAULT_DEPRECATION_CHECK_INTERVAL_SECONDS
|
||||
]
|
||||
mock_send_alert.assert_awaited_once()
|
||||
assert "dead-alias" in mock_send_alert.await_args.kwargs["message"]
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue