test: deflake redis loop-stall burst test and pre-commit interrupt cleanup

The redis breaker test raced the event loop: the fake call had to still be
pending when a real time.sleep stall began, which needs the loop to get from
scheduling to the stall in under 1ms. The fake now holds its answer behind an
asyncio.Event so the whole burst times out deterministically.

The pre-commit interrupt test found a real leak: lint_dashboard creates its
eslint report with mktemp and only removed it on the happy path, so an
interrupt landing during the whole-folder eslint run left the file behind.
The subshell now removes it from an EXIT trap, and the test drives the
interrupt while that eslint run is in flight.

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
mateo 2026-09-04 10:07:34 +00:00
parent 9835883e03
commit 03a82823fc
3 changed files with 17 additions and 12 deletions

View file

@ -142,6 +142,8 @@ fi
lint_dashboard() {
(
trap 'exit 143' TERM
trap 'rm -f "${report:-}"' EXIT
rc=0
prettier_rel=()
eslint_rel=()
@ -168,7 +170,6 @@ EOF
report=$(mktemp)
npx eslint . -f json -o "$report" || true
node scripts/check-lint-budgets.mjs "$report" eslint-budgets.json || rc=1
rm -f "$report"
exit $rc
)
}

View file

@ -822,30 +822,27 @@ async def test_event_loop_stall_timeout_burst_keeps_breaker_closed():
Every operation already waiting on the loop times out together when the loop resumes,
so a purely consecutive threshold is satisfied instantly even though the Redis on the
other end (here an in-process fake that answers immediately) is healthy.
other end is healthy. The stall is modelled by holding the fake Redis's answers back
until the whole burst has hit its client timeout, then releasing them.
"""
import time as time_mod
from litellm.caching.redis_cache import RedisCircuitBreaker, _run_under_circuit_breaker
breaker = RedisCircuitBreaker(failure_threshold=3, recovery_timeout=60, timeout_min_duration=5.0)
loop_resumed = asyncio.Event()
async def healthy_redis_call_with_client_timeout():
return await asyncio.wait_for(asyncio.sleep(0.001, result="ok"), timeout=0.05)
async def stall_the_loop():
await asyncio.sleep(0)
time_mod.sleep(0.2)
await asyncio.wait_for(loop_resumed.wait(), timeout=0.05)
return "ok"
results = await asyncio.gather(
*(_run_under_circuit_breaker(breaker, "op", healthy_redis_call_with_client_timeout) for _ in range(8)),
stall_the_loop(),
return_exceptions=True,
)
timeouts = [r for r in results if isinstance(r, asyncio.TimeoutError)]
assert len(timeouts) >= breaker.failure_threshold, "the stall must time out a full burst"
assert len(timeouts) == 8, "the stall must time out the whole burst"
assert breaker.is_open() is False, "a healthy Redis behind one loop stall must stay in the pool"
loop_resumed.set()
assert await _run_under_circuit_breaker(breaker, "op", healthy_redis_call_with_client_timeout) == "ok"

View file

@ -53,6 +53,12 @@ case "$*" in
"eslint --no-warn-ignored"*)
[ "${STUB_FAIL:-}" = "eslint" ] && exit 1
;;
"eslint . -f json"*)
if [ -n "${STUB_HANG_DIR:-}" ]; then
touch "$STUB_HANG_DIR/eslint_report.started"
sleep 60
fi
;;
esac
exit 0
"""
@ -340,11 +346,12 @@ def test_interrupt_kills_background_jobs_and_removes_logs(tmp_path: Path) -> Non
)
try:
assert _wait_until((hang_dir / "make.started").exists, 10)
assert _wait_until((hang_dir / "eslint_report.started").exists, 10)
os.killpg(proc.pid, signal.SIGINT)
assert proc.wait(timeout=10) != 0
make_pid = int((hang_dir / "make.pid").read_text())
assert _wait_until(lambda: _pid_gone(make_pid), 5)
assert list(tmp_dir.iterdir()) == []
assert _wait_until(lambda: not any(tmp_dir.iterdir()), 5), list(tmp_dir.iterdir())
finally:
with suppress(ProcessLookupError, PermissionError):
os.killpg(proc.pid, signal.SIGTERM)