fix(langfuse): cap LANGFUSE_MAX_RETRIES at 1000 so an absurd value cannot stall callback init

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
yucheng 2026-09-22 19:08:37 +00:00
parent 10f05c63db
commit d6086f0917
2 changed files with 30 additions and 3 deletions

View file

@ -82,6 +82,7 @@ _DEFAULT_FLUSH_AT: Final = 512
_CHANNEL_RETIRE_GRACE_SECONDS: Final = 60.0
_DEFAULT_TIMEOUT_SECONDS: Final = 20.0
_DEFAULT_MAX_RETRIES: Final = 3
_MAX_RETRIES: Final = 1_000
_MAX_BACKOFF_EXPONENT: Final = 6
_DEFAULT_PROMPT_CACHE_TTL_SECONDS: Final = 60.0
_JSON_SAFE_INT: Final = 2**53 - 1
@ -482,7 +483,11 @@ def configured_timeout() -> float:
def configured_max_retries() -> int:
"""``LANGFUSE_MAX_RETRIES`` as the number of re-sends after a failed export, the v2 SDK's knob and default."""
"""``LANGFUSE_MAX_RETRIES`` as the number of re-sends after a failed export, the v2 SDK's knob and default.
Capped at ``_MAX_RETRIES``: with the backoff ceiling that is already hours per batch, and the exporter holds
one delay per re-send.
"""
raw: Final = os.environ.get("LANGFUSE_MAX_RETRIES")
if raw is None:
return _DEFAULT_MAX_RETRIES
@ -491,7 +496,12 @@ def configured_max_retries() -> int:
"LANGFUSE_MAX_RETRIES=%r is not a whole number; retrying %d times", raw, _DEFAULT_MAX_RETRIES
)
return _DEFAULT_MAX_RETRIES
return int(raw)
requested: Final = int(raw)
if requested > _MAX_RETRIES:
verbose_logger.warning(
"LANGFUSE_MAX_RETRIES=%d is above the ceiling; retrying %d times", requested, _MAX_RETRIES
)
return min(requested, _MAX_RETRIES)
def configured_release() -> str | None:

View file

@ -1311,11 +1311,28 @@ def test_large_retry_count_builds_an_exporter_with_capped_backoff(monkeypatch):
and take the whole callback down at init."""
monkeypatch.setenv("LANGFUSE_MAX_RETRIES", "1025")
exporter = _build_span_exporter(public_key="pk", secret_key="sk", base_url="https://lf.internal.example")
assert len(exporter.delays) == 1025
assert 3 < len(exporter.delays) <= 1025
assert exporter.delays[:4] == (1.0, 2.0, 4.0, 8.0)
assert max(exporter.delays) == exporter.delays[-1] <= 64.0
def test_absurd_retry_count_is_clamped_instead_of_allocating_one_delay_per_retry(monkeypatch, caplog):
"""A retry count with twelve digits must not turn callback init into a multi-gigabyte tuple allocation."""
monkeypatch.setenv("LANGFUSE_MAX_RETRIES", "999999999999")
with caplog.at_level(logging.WARNING, logger="LiteLLM"):
exporter = _build_span_exporter(public_key="pk", secret_key="sk", base_url="https://lf.internal.example")
assert 3 < len(exporter.delays) <= 1025
assert exporter.delays[-1] <= 64.0
assert any("LANGFUSE_MAX_RETRIES=999999999999" in record.getMessage() for record in caplog.records)
caplog.clear()
monkeypatch.setenv("LANGFUSE_MAX_RETRIES", "5")
with caplog.at_level(logging.WARNING, logger="LiteLLM"):
modest = _build_span_exporter(public_key="pk", secret_key="sk", base_url="https://lf.internal.example")
assert len(modest.delays) == 5
assert not any("LANGFUSE_MAX_RETRIES" in record.getMessage() for record in caplog.records)
def test_enable_langfuse_debug_logging_makes_deliveries_visible_on_the_langfuse_logger(caplog):
"""``LANGFUSE_DEBUG`` turned on the v2 SDK's own logger; it has to do the same for litellm's export channel."""
exporter, _ = _exporter_over([200])