mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-30 01:52:18 +00:00
fix(langfuse): cap LANGFUSE_MAX_RETRIES at 1000 so an absurd value cannot stall callback init
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
10f05c63db
commit
d6086f0917
2 changed files with 30 additions and 3 deletions
|
|
@ -82,6 +82,7 @@ _DEFAULT_FLUSH_AT: Final = 512
|
|||
_CHANNEL_RETIRE_GRACE_SECONDS: Final = 60.0
|
||||
_DEFAULT_TIMEOUT_SECONDS: Final = 20.0
|
||||
_DEFAULT_MAX_RETRIES: Final = 3
|
||||
_MAX_RETRIES: Final = 1_000
|
||||
_MAX_BACKOFF_EXPONENT: Final = 6
|
||||
_DEFAULT_PROMPT_CACHE_TTL_SECONDS: Final = 60.0
|
||||
_JSON_SAFE_INT: Final = 2**53 - 1
|
||||
|
|
@ -482,7 +483,11 @@ def configured_timeout() -> float:
|
|||
|
||||
|
||||
def configured_max_retries() -> int:
|
||||
"""``LANGFUSE_MAX_RETRIES`` as the number of re-sends after a failed export, the v2 SDK's knob and default."""
|
||||
"""``LANGFUSE_MAX_RETRIES`` as the number of re-sends after a failed export, the v2 SDK's knob and default.
|
||||
|
||||
Capped at ``_MAX_RETRIES``: with the backoff ceiling that is already hours per batch, and the exporter holds
|
||||
one delay per re-send.
|
||||
"""
|
||||
raw: Final = os.environ.get("LANGFUSE_MAX_RETRIES")
|
||||
if raw is None:
|
||||
return _DEFAULT_MAX_RETRIES
|
||||
|
|
@ -491,7 +496,12 @@ def configured_max_retries() -> int:
|
|||
"LANGFUSE_MAX_RETRIES=%r is not a whole number; retrying %d times", raw, _DEFAULT_MAX_RETRIES
|
||||
)
|
||||
return _DEFAULT_MAX_RETRIES
|
||||
return int(raw)
|
||||
requested: Final = int(raw)
|
||||
if requested > _MAX_RETRIES:
|
||||
verbose_logger.warning(
|
||||
"LANGFUSE_MAX_RETRIES=%d is above the ceiling; retrying %d times", requested, _MAX_RETRIES
|
||||
)
|
||||
return min(requested, _MAX_RETRIES)
|
||||
|
||||
|
||||
def configured_release() -> str | None:
|
||||
|
|
|
|||
|
|
@ -1311,11 +1311,28 @@ def test_large_retry_count_builds_an_exporter_with_capped_backoff(monkeypatch):
|
|||
and take the whole callback down at init."""
|
||||
monkeypatch.setenv("LANGFUSE_MAX_RETRIES", "1025")
|
||||
exporter = _build_span_exporter(public_key="pk", secret_key="sk", base_url="https://lf.internal.example")
|
||||
assert len(exporter.delays) == 1025
|
||||
assert 3 < len(exporter.delays) <= 1025
|
||||
assert exporter.delays[:4] == (1.0, 2.0, 4.0, 8.0)
|
||||
assert max(exporter.delays) == exporter.delays[-1] <= 64.0
|
||||
|
||||
|
||||
def test_absurd_retry_count_is_clamped_instead_of_allocating_one_delay_per_retry(monkeypatch, caplog):
|
||||
"""A retry count with twelve digits must not turn callback init into a multi-gigabyte tuple allocation."""
|
||||
monkeypatch.setenv("LANGFUSE_MAX_RETRIES", "999999999999")
|
||||
with caplog.at_level(logging.WARNING, logger="LiteLLM"):
|
||||
exporter = _build_span_exporter(public_key="pk", secret_key="sk", base_url="https://lf.internal.example")
|
||||
assert 3 < len(exporter.delays) <= 1025
|
||||
assert exporter.delays[-1] <= 64.0
|
||||
assert any("LANGFUSE_MAX_RETRIES=999999999999" in record.getMessage() for record in caplog.records)
|
||||
|
||||
caplog.clear()
|
||||
monkeypatch.setenv("LANGFUSE_MAX_RETRIES", "5")
|
||||
with caplog.at_level(logging.WARNING, logger="LiteLLM"):
|
||||
modest = _build_span_exporter(public_key="pk", secret_key="sk", base_url="https://lf.internal.example")
|
||||
assert len(modest.delays) == 5
|
||||
assert not any("LANGFUSE_MAX_RETRIES" in record.getMessage() for record in caplog.records)
|
||||
|
||||
|
||||
def test_enable_langfuse_debug_logging_makes_deliveries_visible_on_the_langfuse_logger(caplog):
|
||||
"""``LANGFUSE_DEBUG`` turned on the v2 SDK's own logger; it has to do the same for litellm's export channel."""
|
||||
exporter, _ = _exporter_over([200])
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue