diff --git a/litellm/integrations/langfuse/langfuse_sdk.py b/litellm/integrations/langfuse/langfuse_sdk.py index 62aeb953f5e..83fc75d13a6 100644 --- a/litellm/integrations/langfuse/langfuse_sdk.py +++ b/litellm/integrations/langfuse/langfuse_sdk.py @@ -82,6 +82,7 @@ _DEFAULT_FLUSH_AT: Final = 512 _CHANNEL_RETIRE_GRACE_SECONDS: Final = 60.0 _DEFAULT_TIMEOUT_SECONDS: Final = 20.0 _DEFAULT_MAX_RETRIES: Final = 3 +_MAX_RETRIES: Final = 1_000 _MAX_BACKOFF_EXPONENT: Final = 6 _DEFAULT_PROMPT_CACHE_TTL_SECONDS: Final = 60.0 _JSON_SAFE_INT: Final = 2**53 - 1 @@ -482,7 +483,11 @@ def configured_timeout() -> float: def configured_max_retries() -> int: - """``LANGFUSE_MAX_RETRIES`` as the number of re-sends after a failed export, the v2 SDK's knob and default.""" + """``LANGFUSE_MAX_RETRIES`` as the number of re-sends after a failed export, the v2 SDK's knob and default. + + Capped at ``_MAX_RETRIES``: with the backoff ceiling that is already hours per batch, and the exporter holds + one delay per re-send. + """ raw: Final = os.environ.get("LANGFUSE_MAX_RETRIES") if raw is None: return _DEFAULT_MAX_RETRIES @@ -491,7 +496,12 @@ def configured_max_retries() -> int: "LANGFUSE_MAX_RETRIES=%r is not a whole number; retrying %d times", raw, _DEFAULT_MAX_RETRIES ) return _DEFAULT_MAX_RETRIES - return int(raw) + requested: Final = int(raw) + if requested > _MAX_RETRIES: + verbose_logger.warning( + "LANGFUSE_MAX_RETRIES=%d is above the ceiling; retrying %d times", requested, _MAX_RETRIES + ) + return min(requested, _MAX_RETRIES) def configured_release() -> str | None: diff --git a/tests/test_litellm/integrations/langfuse/test_langfuse_sdk.py b/tests/test_litellm/integrations/langfuse/test_langfuse_sdk.py index 2603aae6211..dfe5f284226 100644 --- a/tests/test_litellm/integrations/langfuse/test_langfuse_sdk.py +++ b/tests/test_litellm/integrations/langfuse/test_langfuse_sdk.py @@ -1311,11 +1311,28 @@ def test_large_retry_count_builds_an_exporter_with_capped_backoff(monkeypatch): and take the whole callback down at init.""" monkeypatch.setenv("LANGFUSE_MAX_RETRIES", "1025") exporter = _build_span_exporter(public_key="pk", secret_key="sk", base_url="https://lf.internal.example") - assert len(exporter.delays) == 1025 + assert 3 < len(exporter.delays) <= 1025 assert exporter.delays[:4] == (1.0, 2.0, 4.0, 8.0) assert max(exporter.delays) == exporter.delays[-1] <= 64.0 +def test_absurd_retry_count_is_clamped_instead_of_allocating_one_delay_per_retry(monkeypatch, caplog): + """A retry count with twelve digits must not turn callback init into a multi-gigabyte tuple allocation.""" + monkeypatch.setenv("LANGFUSE_MAX_RETRIES", "999999999999") + with caplog.at_level(logging.WARNING, logger="LiteLLM"): + exporter = _build_span_exporter(public_key="pk", secret_key="sk", base_url="https://lf.internal.example") + assert 3 < len(exporter.delays) <= 1025 + assert exporter.delays[-1] <= 64.0 + assert any("LANGFUSE_MAX_RETRIES=999999999999" in record.getMessage() for record in caplog.records) + + caplog.clear() + monkeypatch.setenv("LANGFUSE_MAX_RETRIES", "5") + with caplog.at_level(logging.WARNING, logger="LiteLLM"): + modest = _build_span_exporter(public_key="pk", secret_key="sk", base_url="https://lf.internal.example") + assert len(modest.delays) == 5 + assert not any("LANGFUSE_MAX_RETRIES" in record.getMessage() for record in caplog.records) + + def test_enable_langfuse_debug_logging_makes_deliveries_visible_on_the_langfuse_logger(caplog): """``LANGFUSE_DEBUG`` turned on the v2 SDK's own logger; it has to do the same for litellm's export channel.""" exporter, _ = _exporter_over([200])