fix(proxy): treat Postgres connection exhaustion as DB unavailable, not poison spend-log rows (#44270)

Co-authored-by: yassin <yassin@berri.ai>
Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
devin-ai-integration[bot] 2026-10-03 14:28:52 -07:00 • committed by GitHub
parent e9bf2cfd01
commit 9a4b1951a2
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
3 changed files with 133 additions and 2 deletions

View file

@ -24,6 +24,12 @@ _TRANSIENT_DB_UNAVAILABLE_MESSAGE: Final = (
_DATABASE_ERROR_META: Final = TypeAdapter(dict[str, object])
_BATCH_POSTGRES_ERROR_CODE: Final = re.compile(r'PostgresError \{ code: "([0-9A-Z]{5})"')
_CONNECTION_CAPACITY_PHRASES: Final = (
"too many clients already",
"too many connections for role",
"too many connections for database",
"remaining connection slots are reserved",
)
def _exception_chain(e: BaseException) -> Iterator[BaseException]:
@ -236,6 +242,22 @@ class PrismaDBExceptionHandler:
or "write conflict or a deadlock" in error_message
)
@staticmethod
def is_database_capacity_error(e: Exception) -> bool:
"""True iff Postgres refused the pool a new connection (SQLSTATE 53300:
``too many clients already``, a reserved slot, or a per-role limit). The
server is up but full, so the failure is neither a transport error (which
would tear the engine down and open yet more connections against it) nor
a rejection of the rows being written."""
import prisma
if not isinstance(e, _exception_types(prisma.errors.PrismaError)):
return False
if PrismaDBExceptionHandler.postgres_sqlstate(e) == "53300":
return True
error_message: Final = str(e).lower()
return any(phrase in error_message for phrase in _CONNECTION_CAPACITY_PHRASES)
@staticmethod
def postgres_sqlstate(e: Exception) -> str | None:
"""The SQLSTATE Postgres attached to a failed statement, as prisma surfaces it, or None."""
@ -325,6 +347,8 @@ class PrismaDBExceptionHandler:
return True
if PrismaDBExceptionHandler.is_database_transport_error(e):
return True
if PrismaDBExceptionHandler.is_database_capacity_error(e):
return True
if PrismaDBExceptionHandler.is_prisma_engine_internal_error(e):
return True
if "cached plan must not change result type" in str(e).lower():

View file

@ -789,3 +789,75 @@ def test_db_lookup_deadline_is_a_connection_and_unavailability_error_but_never_a
assert PrismaDBExceptionHandler.is_database_service_unavailable_error(deadline) is True
assert PrismaDBExceptionHandler.is_database_transport_error(deadline) is False
assert "temporarily unreachable" in PrismaDBExceptionHandler.database_unavailable_message(deadline)
_TOO_MANY_CLIENTS: Final = "Error in connector: Error querying the database: FATAL: sorry, too many clients already"
@pytest.mark.parametrize(
"error",
[
DataError(data={"user_facing_error": {"message": _TOO_MANY_CLIENTS}}),
DataError(
data={
"user_facing_error": {
"message": "Error querying the database: FATAL: zu viele Verbindungen",
"meta": {"code": "53300", "message": "zu viele Verbindungen"},
}
}
),
DataError(
data={
"user_facing_error": {
"message": "Error in connector: Error querying the database: FATAL: remaining connection slots are reserved for roles with the SUPERUSER attribute"
}
}
),
DataError(
data={
"user_facing_error": {
"message": 'Error occurred during query execution: ConnectorError(ConnectorError { user_facing_error: None, kind: QueryError(PostgresError { code: "53300", message: "too many connections for role \\"litellm\\"", severity: "FATAL" }) })'
}
}
),
DataError(
data={
"user_facing_error": {
"message": 'Error in connector: Error querying the database: FATAL: too many connections for role "litellm"'
}
}
),
DataError(
data={
"user_facing_error": {
"message": 'Error in connector: Error querying the database: FATAL: too many connections for database "litellm"'
}
}
),
],
)
def test_postgres_connection_capacity_refusal_is_service_unavailable_not_a_data_error(error: DataError) -> None:
"""Postgres refusing a new connection (SQLSTATE 53300) reaches the proxy as a
bare ``DataError`` with no SQLSTATE in ``meta``. The server is up but full, so
the failure is service-unavailable (the spend-log flush re-raises and requeues
instead of bisecting the batch row by row against a full server) while not a
transport error, which would make auth and the health check tear the engine
down and open yet more connections against it."""
assert PrismaDBExceptionHandler.is_database_capacity_error(error) is True
assert PrismaDBExceptionHandler.is_database_service_unavailable_error(error) is True
assert PrismaDBExceptionHandler.is_database_transport_error(error) is False
assert PrismaDBExceptionHandler.is_prisma_data_error(error) is True
@pytest.mark.parametrize(
"error",
[
DataError(data={"user_facing_error": {"message": "invalid byte sequence for encoding UTF8: 0x00"}}),
UniqueViolationError(data={"user_facing_error": {"error_code": "P2002", "meta": {"table": "t"}}}),
PrismaError("can't reach database server"),
httpx.ConnectError("connection refused"),
RuntimeError(_TOO_MANY_CLIENTS),
],
)
def test_is_database_capacity_error_excludes_other_failures(error: Exception) -> None:
assert PrismaDBExceptionHandler.is_database_capacity_error(error) is False

View file

@ -10,8 +10,8 @@ from __future__ import annotations
import asyncio
import json
from collections.abc import Iterator
from typing import Any, Dict, List
from collections.abc import Callable, Iterator
from typing import Any, Dict, Final, List
from unittest.mock import AsyncMock, MagicMock
import pytest
@ -917,3 +917,38 @@ async def test_update_spend_logs_parks_failed_batch_in_redis_with_wire_safe_date
parked = await buffer.get_spend_logs_from_redis_buffer(limit=10)
assert mock_prisma_client.spend_log_transactions == []
assert [(row["request_id"], row["startTime"]) for row in parked] == [("a", started.isoformat())]
@pytest.mark.asyncio
async def test_update_spend_logs_requeues_batch_when_postgres_is_out_of_connections(
mock_prisma_client: MagicMock,
make_spend_log_row: Callable[..., Dict[str, Any]],
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""Postgres refusing the pool a new connection (SQLSTATE 53300, "too many
clients already") surfaces as a bare ``DataError``. It is the server being
full, not a row being bad: the flush must stop after the one failed insert
and put the batch back at the head of the queue for the next interval,
rather than bisecting it (hundreds more connection attempts against a full
server) and dropping every row as poisoned."""
sleep: Final = AsyncMock(return_value=None)
monkeypatch.setattr(utils_mod.asyncio, "sleep", sleep)
err: Final = _data_error("Error in connector: Error querying the database: FATAL: sorry, too many clients already")
mock_prisma_client.db.litellm_spendlogs.create_many = AsyncMock(side_effect=err)
proxy_logging: Final = MagicMock()
proxy_logging.failure_handler = AsyncMock()
mock_prisma_client.spend_log_transactions = [make_spend_log_row(request_id="e")]
logs: Final = [make_spend_log_row(request_id=f"r{i}") for i in range(4)]
with pytest.raises(type(err)):
await ProxyUpdateSpend.update_spend_logs(
n_retry_times=2,
prisma_client=mock_prisma_client,
db_writer_client=None,
proxy_logging_obj=proxy_logging,
logs_to_process=logs,
)
assert mock_prisma_client.db.litellm_spendlogs.create_many.await_count == 1
sleep.assert_not_awaited()
assert [row["request_id"] for row in mock_prisma_client.spend_log_transactions] == ["r0", "r1", "r2", "r3", "e"]