fix(prisma): reset reconnect failure counter on successful watchdog probe

_consecutive_reconnect_failures was only reset inside _attempt_reconnect_inside_lock
on a successful reconnect cycle. A successful SELECT 1 watchdog probe, which is
the steady-state signal that the engine is healthy, never touched the counter.
This left it permanently elevated after any transient reconnect failure,
causing the next probe failure to escalate to the heavy reconnect path
even when the engine had been serving traffic normally for an extended period.
This commit is contained in:
gvisco 2026-06-02 15:53:04 +02:00
parent f48a87ef12
commit c2cd182c0c
2 changed files with 22 additions and 0 deletions

View file

@ -4775,6 +4775,7 @@ class PrismaClient:
self.db.query_raw("SELECT 1"),
timeout=self._db_health_watchdog_probe_timeout_seconds,
)
self._consecutive_reconnect_failures = 0
except asyncio.CancelledError:
break
except Exception as e:

View file

@ -486,3 +486,24 @@ async def test_engine_confirmed_dead_persists_across_failed_heavy_reconnect(
# The flag must STILL be True so the next attempt re-enters the heavy
# branch instead of silently demoting to the lightweight path.
assert client._engine_confirmed_dead is True
@pytest.mark.asyncio
async def test_db_health_watchdog_resets_failure_counter_on_successful_probe(
mock_proxy_logging,
):
client = PrismaClient(
database_url="mock://test", proxy_logging_obj=mock_proxy_logging
)
client.db.query_raw = AsyncMock(return_value=[{"result": 1}])
client._consecutive_reconnect_failures = 3
client._db_health_watchdog_interval_seconds = 1
client._db_health_watchdog_probe_timeout_seconds = 0.2
with patch(
"litellm.proxy.utils.asyncio.sleep",
AsyncMock(side_effect=[None, asyncio.CancelledError()]),
):
await client._db_health_watchdog_loop()
assert client._consecutive_reconnect_failures == 0