mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-11 03:38:38 +00:00
fix(prometheus): ignore a non-positive series cap or TTL with a warning instead of failing the logger
A cap or TTL of 0 or less raised at logger init. The proxy logs that as a non-blocking error and keeps serving, so the result was a running proxy with no Prometheus metrics at all. The setting is now ignored with a startup warning naming it, the same rule the end_user cap already follows for a non-positive value
This commit is contained in:
parent
2aefd9c76e
commit
77dd85c663
3 changed files with 31 additions and 13 deletions
|
|
@ -255,6 +255,19 @@ class _LabeledMetric:
|
|||
|
||||
_MetricLike: TypeAlias = "NoOpMetric | _LabeledMetric | MetricWrapperBase"
|
||||
|
||||
_SeriesLimitT: Final = TypeVar("_SeriesLimitT", int, float)
|
||||
|
||||
|
||||
def _positive_or_ignored(setting: str, value: _SeriesLimitT | None) -> _SeriesLimitT | None:
|
||||
if value is None or value > 0:
|
||||
return value
|
||||
verbose_logger.warning(
|
||||
"%s is ignored because it is not greater than 0 (got %s). Prometheus metrics are emitted without it",
|
||||
setting,
|
||||
value,
|
||||
)
|
||||
return None
|
||||
|
||||
|
||||
def _get_budget_metrics_per_request_timeout() -> float:
|
||||
raw: Final = os.getenv("PROMETHEUS_BUDGET_METRICS_PER_REQUEST_TIMEOUT")
|
||||
|
|
@ -1285,8 +1298,10 @@ class PrometheusLogger(CustomLogger):
|
|||
@staticmethod
|
||||
def _configured_series_limits(multiprocess_mode: bool) -> PrometheusSeriesLimits:
|
||||
limits: Final = PrometheusSeriesLimits(
|
||||
max_series=litellm.prometheus_metrics_max_series_per_metric,
|
||||
ttl_seconds=litellm.prometheus_metrics_ttl_seconds,
|
||||
max_series=_positive_or_ignored(
|
||||
"prometheus_metrics_max_series_per_metric", litellm.prometheus_metrics_max_series_per_metric
|
||||
),
|
||||
ttl_seconds=_positive_or_ignored("prometheus_metrics_ttl_seconds", litellm.prometheus_metrics_ttl_seconds),
|
||||
cleanup_interval_seconds=litellm.prometheus_metrics_cleanup_interval_seconds,
|
||||
)
|
||||
if limits.ttl_seconds is None or not multiprocess_mode:
|
||||
|
|
|
|||
|
|
@ -19,14 +19,6 @@ class PrometheusSeriesLimits:
|
|||
ttl_seconds: float | None
|
||||
cleanup_interval_seconds: float | None
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
if self.max_series is not None and self.max_series <= 0:
|
||||
raise ValueError(
|
||||
f"prometheus_metrics_max_series_per_metric must be a positive integer, got {self.max_series}"
|
||||
)
|
||||
if self.ttl_seconds is not None and self.ttl_seconds <= 0:
|
||||
raise ValueError(f"prometheus_metrics_ttl_seconds must be a positive number, got {self.ttl_seconds}")
|
||||
|
||||
@property
|
||||
def enabled(self) -> bool:
|
||||
return self.max_series is not None or self.ttl_seconds is not None
|
||||
|
|
|
|||
|
|
@ -1,3 +1,4 @@
|
|||
import logging
|
||||
import re
|
||||
from pathlib import Path
|
||||
from threading import Thread
|
||||
|
|
@ -368,8 +369,18 @@ def test_series_stay_unbounded_unless_a_limit_is_configured():
|
|||
("prometheus_metrics_ttl_seconds", -1.0),
|
||||
],
|
||||
)
|
||||
def test_non_positive_series_limits_fail_logger_startup(setting: str, value: float):
|
||||
def test_a_non_positive_series_limit_is_ignored_with_a_warning_and_metrics_keep_flowing(
|
||||
setting: str, value: float, clock, caplog
|
||||
):
|
||||
litellm.prometheus_metrics_cleanup_interval_seconds = 0.0
|
||||
setattr(litellm, setting, value)
|
||||
|
||||
with pytest.raises(ValueError, match=setting):
|
||||
PrometheusLogger()
|
||||
with caplog.at_level(logging.WARNING, logger="LiteLLM"):
|
||||
logger: Final = PrometheusLogger()
|
||||
for index in range(3):
|
||||
_count_request(logger, f"agent-{index}")
|
||||
clock[0] += 100.0
|
||||
|
||||
series: Final = _scraped_series("litellm_proxy_total_requests_metric_total")
|
||||
assert _label_values(series, "user_agent") == {"agent-0", "agent-1", "agent-2"}
|
||||
assert setting in caplog.text
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue