diff --git a/litellm/proxy/db/exception_handler.py b/litellm/proxy/db/exception_handler.py index 022ca2efc9e..7eb8f564225 100644 --- a/litellm/proxy/db/exception_handler.py +++ b/litellm/proxy/db/exception_handler.py @@ -24,6 +24,12 @@ _TRANSIENT_DB_UNAVAILABLE_MESSAGE: Final = ( _DATABASE_ERROR_META: Final = TypeAdapter(dict[str, object]) _BATCH_POSTGRES_ERROR_CODE: Final = re.compile(r'PostgresError \{ code: "([0-9A-Z]{5})"') +_CONNECTION_CAPACITY_PHRASES: Final = ( + "too many clients already", + "too many connections for role", + "too many connections for database", + "remaining connection slots are reserved", +) def _exception_chain(e: BaseException) -> Iterator[BaseException]: @@ -236,6 +242,22 @@ class PrismaDBExceptionHandler: or "write conflict or a deadlock" in error_message ) + @staticmethod + def is_database_capacity_error(e: Exception) -> bool: + """True iff Postgres refused the pool a new connection (SQLSTATE 53300: + ``too many clients already``, a reserved slot, or a per-role limit). The + server is up but full, so the failure is neither a transport error (which + would tear the engine down and open yet more connections against it) nor + a rejection of the rows being written.""" + import prisma + + if not isinstance(e, _exception_types(prisma.errors.PrismaError)): + return False + if PrismaDBExceptionHandler.postgres_sqlstate(e) == "53300": + return True + error_message: Final = str(e).lower() + return any(phrase in error_message for phrase in _CONNECTION_CAPACITY_PHRASES) + @staticmethod def postgres_sqlstate(e: Exception) -> str | None: """The SQLSTATE Postgres attached to a failed statement, as prisma surfaces it, or None.""" @@ -325,6 +347,8 @@ class PrismaDBExceptionHandler: return True if PrismaDBExceptionHandler.is_database_transport_error(e): return True + if PrismaDBExceptionHandler.is_database_capacity_error(e): + return True if PrismaDBExceptionHandler.is_prisma_engine_internal_error(e): return True if "cached plan must not change result type" in str(e).lower(): diff --git a/tests/unit/proxy/db/test_exception_handler.py b/tests/unit/proxy/db/test_exception_handler.py index 09f4d294ad0..c00deea7de5 100644 --- a/tests/unit/proxy/db/test_exception_handler.py +++ b/tests/unit/proxy/db/test_exception_handler.py @@ -789,3 +789,75 @@ def test_db_lookup_deadline_is_a_connection_and_unavailability_error_but_never_a assert PrismaDBExceptionHandler.is_database_service_unavailable_error(deadline) is True assert PrismaDBExceptionHandler.is_database_transport_error(deadline) is False assert "temporarily unreachable" in PrismaDBExceptionHandler.database_unavailable_message(deadline) + + +_TOO_MANY_CLIENTS: Final = "Error in connector: Error querying the database: FATAL: sorry, too many clients already" + + +@pytest.mark.parametrize( + "error", + [ + DataError(data={"user_facing_error": {"message": _TOO_MANY_CLIENTS}}), + DataError( + data={ + "user_facing_error": { + "message": "Error querying the database: FATAL: zu viele Verbindungen", + "meta": {"code": "53300", "message": "zu viele Verbindungen"}, + } + } + ), + DataError( + data={ + "user_facing_error": { + "message": "Error in connector: Error querying the database: FATAL: remaining connection slots are reserved for roles with the SUPERUSER attribute" + } + } + ), + DataError( + data={ + "user_facing_error": { + "message": 'Error occurred during query execution: ConnectorError(ConnectorError { user_facing_error: None, kind: QueryError(PostgresError { code: "53300", message: "too many connections for role \\"litellm\\"", severity: "FATAL" }) })' + } + } + ), + DataError( + data={ + "user_facing_error": { + "message": 'Error in connector: Error querying the database: FATAL: too many connections for role "litellm"' + } + } + ), + DataError( + data={ + "user_facing_error": { + "message": 'Error in connector: Error querying the database: FATAL: too many connections for database "litellm"' + } + } + ), + ], +) +def test_postgres_connection_capacity_refusal_is_service_unavailable_not_a_data_error(error: DataError) -> None: + """Postgres refusing a new connection (SQLSTATE 53300) reaches the proxy as a + bare ``DataError`` with no SQLSTATE in ``meta``. The server is up but full, so + the failure is service-unavailable (the spend-log flush re-raises and requeues + instead of bisecting the batch row by row against a full server) while not a + transport error, which would make auth and the health check tear the engine + down and open yet more connections against it.""" + assert PrismaDBExceptionHandler.is_database_capacity_error(error) is True + assert PrismaDBExceptionHandler.is_database_service_unavailable_error(error) is True + assert PrismaDBExceptionHandler.is_database_transport_error(error) is False + assert PrismaDBExceptionHandler.is_prisma_data_error(error) is True + + +@pytest.mark.parametrize( + "error", + [ + DataError(data={"user_facing_error": {"message": "invalid byte sequence for encoding UTF8: 0x00"}}), + UniqueViolationError(data={"user_facing_error": {"error_code": "P2002", "meta": {"table": "t"}}}), + PrismaError("can't reach database server"), + httpx.ConnectError("connection refused"), + RuntimeError(_TOO_MANY_CLIENTS), + ], +) +def test_is_database_capacity_error_excludes_other_failures(error: Exception) -> None: + assert PrismaDBExceptionHandler.is_database_capacity_error(error) is False diff --git a/tests/unit/proxy/utils/prisma_and_spend/test_proxy_update_spend.py b/tests/unit/proxy/utils/prisma_and_spend/test_proxy_update_spend.py index 7099101db1c..fc3deb0de18 100644 --- a/tests/unit/proxy/utils/prisma_and_spend/test_proxy_update_spend.py +++ b/tests/unit/proxy/utils/prisma_and_spend/test_proxy_update_spend.py @@ -10,8 +10,8 @@ from __future__ import annotations import asyncio import json -from collections.abc import Iterator -from typing import Any, Dict, List +from collections.abc import Callable, Iterator +from typing import Any, Dict, Final, List from unittest.mock import AsyncMock, MagicMock import pytest @@ -917,3 +917,38 @@ async def test_update_spend_logs_parks_failed_batch_in_redis_with_wire_safe_date parked = await buffer.get_spend_logs_from_redis_buffer(limit=10) assert mock_prisma_client.spend_log_transactions == [] assert [(row["request_id"], row["startTime"]) for row in parked] == [("a", started.isoformat())] + + +@pytest.mark.asyncio +async def test_update_spend_logs_requeues_batch_when_postgres_is_out_of_connections( + mock_prisma_client: MagicMock, + make_spend_log_row: Callable[..., Dict[str, Any]], + monkeypatch: pytest.MonkeyPatch, +) -> None: + """Postgres refusing the pool a new connection (SQLSTATE 53300, "too many + clients already") surfaces as a bare ``DataError``. It is the server being + full, not a row being bad: the flush must stop after the one failed insert + and put the batch back at the head of the queue for the next interval, + rather than bisecting it (hundreds more connection attempts against a full + server) and dropping every row as poisoned.""" + sleep: Final = AsyncMock(return_value=None) + monkeypatch.setattr(utils_mod.asyncio, "sleep", sleep) + err: Final = _data_error("Error in connector: Error querying the database: FATAL: sorry, too many clients already") + mock_prisma_client.db.litellm_spendlogs.create_many = AsyncMock(side_effect=err) + proxy_logging: Final = MagicMock() + proxy_logging.failure_handler = AsyncMock() + mock_prisma_client.spend_log_transactions = [make_spend_log_row(request_id="e")] + logs: Final = [make_spend_log_row(request_id=f"r{i}") for i in range(4)] + + with pytest.raises(type(err)): + await ProxyUpdateSpend.update_spend_logs( + n_retry_times=2, + prisma_client=mock_prisma_client, + db_writer_client=None, + proxy_logging_obj=proxy_logging, + logs_to_process=logs, + ) + + assert mock_prisma_client.db.litellm_spendlogs.create_many.await_count == 1 + sleep.assert_not_awaited() + assert [row["request_id"] for row in mock_prisma_client.spend_log_transactions] == ["r0", "r1", "r2", "r3", "e"]