fix(db): let the writer pin yield to the replica while the writer is degraded

This commit is contained in:
mateo-berri 2026-08-29 11:44:45 -07:00
parent a8c36e8307
commit 6d3e687ce4
4 changed files with 66 additions and 3 deletions

View file

@ -62,18 +62,23 @@ class _RoutedActions:
class WriterPinnedClient:
"""PrismaClient-shaped view whose `.db` always resolves to the writer.
"""PrismaClient-shaped view whose `.db` resolves to the writer while it is available.
Read-after-write paths (e.g. the model reconcile a /model/new triggers to
verify its own just-committed row) must not read through a lagging read
replica: the row is not replayed there yet, so the reconcile concludes the
write is missing and fails the request even though it is durable (#38556).
While the writer is degraded (`writer_unavailable`), the pin yields to the
routed wrapper so reconcile reads keep working from the replica: a proxy
that starts during a primary outage must still load DB-backed models, and
no read-after-write hazard exists then because writes are failing anyway.
"""
__slots__ = ("db",)
def __init__(self, db: "PrismaWrapper | RoutingPrismaWrapper") -> None:
self.db: Final = db.writer if isinstance(db, RoutingPrismaWrapper) else db
self.db: Final = db.writer if isinstance(db, RoutingPrismaWrapper) and not db.writer_unavailable else db
class RoutingPrismaWrapper:

View file

@ -6769,7 +6769,8 @@ class ProxyConfig:
model write just committed, and reading it through a lagging read replica
makes the write-triggered reload report its own durable write as missing
(#38556). It also keeps a stale replica snapshot from evicting a deployment
another pod just added.
another pod just added. While the writer is degraded the pin yields to the
replica so reader-only mode keeps loading DB-backed models.
"""
try:
new_models: Final[Sequence[_ProxyModelRow]] = await ModelRepository(

View file

@ -126,6 +126,24 @@ def test_writer_pinned_client_passes_through_single_db():
assert WriterPinnedClient(writer).db is writer
def test_writer_pinned_client_yields_to_routed_reads_when_writer_down():
"""The pin must not break reader-only degraded mode: a proxy that starts
during a primary outage still loads DB-backed models from the replica, so
while the writer is degraded the pin resolves to the routed wrapper."""
from litellm.proxy.db.routing_prisma_wrapper import RoutingPrismaWrapper, WriterPinnedClient
writer, writer_inner, reader, reader_inner = _make_wrappers()
writer_inner.litellm_proxymodeltable = _model_actions_mock("writer_models")
reader_inner.litellm_proxymodeltable = _model_actions_mock("reader_models")
routing = RoutingPrismaWrapper(writer=writer, reader=reader)
routing._writer_unavailable = True
pinned = WriterPinnedClient(routing)
assert pinned.db is routing
assert pinned.db.litellm_proxymodeltable.find_many is reader_inner.litellm_proxymodeltable.find_many
@pytest.mark.asyncio
async def test_connect_invokes_both_clients():
from litellm.proxy.db.routing_prisma_wrapper import RoutingPrismaWrapper

View file

@ -9596,6 +9596,45 @@ class TestDeleteDeploymentSync:
assert result == [committed_row], f"Expected the writer's just-committed row, got {result!r}"
reader_inner.litellm_proxymodeltable.find_many.assert_not_awaited()
@pytest.mark.asyncio
async def test_get_models_from_db_falls_back_to_replica_when_writer_down(self):
"""
The writer pin must not break reader-only degraded mode: a proxy that
starts during a primary outage (writer connect failed, replica healthy)
must still load DB-backed models through the replica instead of sending
the reconcile read to the unavailable writer.
"""
from types import SimpleNamespace
from unittest.mock import AsyncMock, MagicMock
from litellm.proxy.db.prisma_client import PrismaWrapper
from litellm.proxy.db.routing_prisma_wrapper import RoutingPrismaWrapper
from litellm.proxy.proxy_server import ProxyConfig
writer_inner = MagicMock(name="writer_prisma")
reader_inner = MagicMock(name="reader_prisma")
replica_row = MagicMock(name="replica_model_row")
writer_inner.litellm_proxymodeltable = SimpleNamespace(
find_many=AsyncMock(side_effect=RuntimeError("writer unreachable")),
create=MagicMock(name="writer_create"),
)
reader_inner.litellm_proxymodeltable = SimpleNamespace(
find_many=AsyncMock(return_value=[replica_row]),
create=MagicMock(name="reader_create"),
)
mock_prisma = MagicMock()
mock_prisma.db = RoutingPrismaWrapper(
writer=PrismaWrapper(original_prisma=writer_inner, iam_token_db_auth=False),
reader=PrismaWrapper(original_prisma=reader_inner, iam_token_db_auth=False),
)
mock_prisma.db._writer_unavailable = True
result = await ProxyConfig()._get_models_from_db(prisma_client=mock_prisma)
assert result == [replica_row], f"Expected the replica's rows in degraded mode, got {result!r}"
writer_inner.litellm_proxymodeltable.find_many.assert_not_awaited()
def test_get_config_list_includes_cancel_on_disconnect(monkeypatch):
"""Follow-up to #30223: the flag must be discoverable via /config/list,