From b735e32ff60afca2e0c20873d707c6d25c233835 Mon Sep 17 00:00:00 2001 From: Ishaan Jaffer Date: Wed, 6 May 2026 16:25:03 -0700 Subject: [PATCH] fix(cleanup): scope dead-daemon update_many to successfully-terminated sessions only MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Greptile P1 (review #PRR_kwDOKALCgc78v9wy): The bulk update_many at the end of _sweep_dead_daemons used the full 'rows' list as the ID set, even for sessions where _terminate_session_internal had thrown. The outer try/except swallowed the error, but the row was still in 'ids' and got marked terminal ('error') without provider.terminate ever firing. Once Epic B wires in a real EC2 provider, that's a silent VM/cost leak per failed sweep. Track terminated_ids inside the try block instead — only successfully terminated rows get the bulk status update. Failed rows stay non-terminal so the next sweep pass retries them. Also early-return when no rows succeed, so we don't fire an empty update_many. --- .../proxy/agent_session_endpoints/cleanup.py | 21 +++++++++++++++---- 1 file changed, 17 insertions(+), 4 deletions(-) diff --git a/litellm/proxy/agent_session_endpoints/cleanup.py b/litellm/proxy/agent_session_endpoints/cleanup.py index 3322670625a..833a5726ac5 100644 --- a/litellm/proxy/agent_session_endpoints/cleanup.py +++ b/litellm/proxy/agent_session_endpoints/cleanup.py @@ -92,9 +92,17 @@ async def _sweep_dead_daemons(prisma_client) -> int: # Terminate via the shared helper. It cancels active runs (with # `run_cancelled` events), flips the session status, AND fires # ``provider.terminate`` — exactly what we want here. + # + # We track ONLY sessions that completed termination cleanly. Failed + # rows stay non-terminal so the next sweep pass can retry them — + # without that, a transient DB error would mark the row ``error`` + # via the bulk update_many below and silently orphan its VM forever + # (Greptile P1). + terminated_ids: list = [] for row in rows: try: await _terminate_session_internal(row.id, reason="daemon_dead") + terminated_ids.append(row.id) except Exception as exc: verbose_proxy_logger.exception( "sweeper: failed to terminate dead-daemon session=%s: %s", @@ -102,13 +110,17 @@ async def _sweep_dead_daemons(prisma_client) -> int: exc, ) + if not terminated_ids: + return 0 + # Daemon-dead sessions land in ``error`` (not ``terminated``); the # shared helper sets ``terminated`` so we explicitly downgrade to # ``error`` here. (Both are terminal — no further state transitions.) + # Scope the bulk update to ONLY successfully-terminated rows so a + # mid-flight failure doesn't get marked terminal-without-VM-teardown. now = _now() - ids = [r.id for r in rows] await prisma_client.db.litellm_agentsession.update_many( - where={"id": {"in": ids}}, + where={"id": {"in": terminated_ids}}, data={ "status": SESSION_STATUS_ERROR, "updated_at": now, @@ -116,9 +128,10 @@ async def _sweep_dead_daemons(prisma_client) -> int: ) verbose_proxy_logger.info( - "sweeper: terminated %d sessions (dead daemon, via provider)", len(ids) + "sweeper: terminated %d sessions (dead daemon, via provider)", + len(terminated_ids), ) - return len(ids) + return len(terminated_ids) async def _sweep_stuck_runs(prisma_client) -> int: