fix(proxy): retry a failed cleanup schedule only when its settings change

_apply_retention_settings rescheduled whenever retention was set and no job existed, so an unparseable cleanup cron was retried on every config reload. Remember the last attempted retention, cron and interval tuple and retry only when it differs

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
yucheng 2026-09-24 18:05:21 +00:00
parent 74fca4514f
commit 755f3e4042
2 changed files with 31 additions and 1 deletions

View file

@ -5081,6 +5081,7 @@ class ProxyConfig:
self._last_websearch_interception_config: dict[str, object] | None = None
self._last_hashicorp_vault_config: dict[str, object] | None = None
self._last_cyberark_config: dict[str, object] | None = None # mutable-ok: change-detection cache
self._last_cleanup_schedule_attempt: tuple[SettingsJsonValue | None, ...] | None = None
self._cyberark_boot_env: dict[str, str | None] | None = None # mutable-ok: deployment env snapshot, set once
self.worker_registry: list[WorkerRegistryEntry] = []
self.config_sync_subscriber: ConfigSyncSubscriber | None = None
@ -7643,7 +7644,14 @@ class ProxyConfig:
resolved: Final = self._resolved_retention_values()
wants_job: Final = any(value is not None for value in resolved)
has_job: Final = scheduler is not None and scheduler.get_job("spend_log_cleanup_job") is not None
if previous_retention_values != resolved or wants_job != has_job:
attempt: Final = (
*resolved,
self.settings.get("maximum_spend_logs_cleanup_cron"),
self.settings.get("maximum_spend_logs_retention_interval"),
)
job_missing: Final = wants_job and not has_job and attempt != self._last_cleanup_schedule_attempt
if previous_retention_values != resolved or job_missing or (has_job and not wants_job):
self._last_cleanup_schedule_attempt = attempt
await self._reschedule_spend_log_cleanup_job()
async def _apply_ssrf_settings(self, db_values: Mapping[str, SettingsJsonValue]) -> None:

View file

@ -3918,6 +3918,28 @@ async def test_ProxyConfig__update_general_settings_schedules_cleanup_when_db_ro
assert fake_scheduler.add_job.call_args.kwargs["id"] == "spend_log_cleanup_job"
@pytest.mark.asyncio
async def test_ProxyConfig__update_general_settings_retries_a_failed_schedule_once_per_settings_value(monkeypatch):
"""An unparseable cron leaves no job behind; reloads must not retry it every tick, only when the
cron or a retention value changes."""
fake_scheduler = MagicMock()
fake_scheduler.get_job.return_value = None
monkeypatch.setattr("litellm.proxy.proxy_server.scheduler", fake_scheduler)
monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", None)
pc = ProxyConfig()
bad_cron = {"maximum_daily_tag_spend_retention_period": "90d", "maximum_spend_logs_cleanup_cron": "not a cron"}
pc.settings.apply_db_row("general_settings", bad_cron)
monkeypatch.setattr("litellm.proxy.proxy_server.general_settings", pc.settings)
for _ in range(3):
await pc._update_general_settings(bad_cron)
assert fake_scheduler.add_job.call_count == 0
assert fake_scheduler.remove_job.call_count == 1
await pc._update_general_settings({**bad_cron, "maximum_spend_logs_cleanup_cron": "* * * * *"})
assert fake_scheduler.add_job.call_count == 1
assert fake_scheduler.add_job.call_args.kwargs["id"] == "spend_log_cleanup_job"
# ---------------------------------------------------------------------------
# ProxyConfig._update_general_settings
# ---------------------------------------------------------------------------