fix(prometheus): ignore a non-positive series cap or TTL with a warning instead of failing the logger

A cap or TTL of 0 or less raised at logger init. The proxy logs that as a non-blocking error and keeps serving, so the result was a running proxy with no Prometheus metrics at all. The setting is now ignored with a startup warning naming it, the same rule the end_user cap already follows for a non-positive value
This commit is contained in:
mateo-berri 2026-10-03 14:48:43 -07:00
parent 2aefd9c76e
commit 77dd85c663
3 changed files with 31 additions and 13 deletions

View file

@ -255,6 +255,19 @@ class _LabeledMetric:
_MetricLike: TypeAlias = "NoOpMetric | _LabeledMetric | MetricWrapperBase"
_SeriesLimitT: Final = TypeVar("_SeriesLimitT", int, float)
def _positive_or_ignored(setting: str, value: _SeriesLimitT | None) -> _SeriesLimitT | None:
if value is None or value > 0:
return value
verbose_logger.warning(
"%s is ignored because it is not greater than 0 (got %s). Prometheus metrics are emitted without it",
setting,
value,
)
return None
def _get_budget_metrics_per_request_timeout() -> float:
raw: Final = os.getenv("PROMETHEUS_BUDGET_METRICS_PER_REQUEST_TIMEOUT")
@ -1285,8 +1298,10 @@ class PrometheusLogger(CustomLogger):
@staticmethod
def _configured_series_limits(multiprocess_mode: bool) -> PrometheusSeriesLimits:
limits: Final = PrometheusSeriesLimits(
max_series=litellm.prometheus_metrics_max_series_per_metric,
ttl_seconds=litellm.prometheus_metrics_ttl_seconds,
max_series=_positive_or_ignored(
"prometheus_metrics_max_series_per_metric", litellm.prometheus_metrics_max_series_per_metric
),
ttl_seconds=_positive_or_ignored("prometheus_metrics_ttl_seconds", litellm.prometheus_metrics_ttl_seconds),
cleanup_interval_seconds=litellm.prometheus_metrics_cleanup_interval_seconds,
)
if limits.ttl_seconds is None or not multiprocess_mode:

View file

@ -19,14 +19,6 @@ class PrometheusSeriesLimits:
ttl_seconds: float | None
cleanup_interval_seconds: float | None
def __post_init__(self) -> None:
if self.max_series is not None and self.max_series <= 0:
raise ValueError(
f"prometheus_metrics_max_series_per_metric must be a positive integer, got {self.max_series}"
)
if self.ttl_seconds is not None and self.ttl_seconds <= 0:
raise ValueError(f"prometheus_metrics_ttl_seconds must be a positive number, got {self.ttl_seconds}")
@property
def enabled(self) -> bool:
return self.max_series is not None or self.ttl_seconds is not None

View file

@ -1,3 +1,4 @@
import logging
import re
from pathlib import Path
from threading import Thread
@ -368,8 +369,18 @@ def test_series_stay_unbounded_unless_a_limit_is_configured():
("prometheus_metrics_ttl_seconds", -1.0),
],
)
def test_non_positive_series_limits_fail_logger_startup(setting: str, value: float):
def test_a_non_positive_series_limit_is_ignored_with_a_warning_and_metrics_keep_flowing(
setting: str, value: float, clock, caplog
):
litellm.prometheus_metrics_cleanup_interval_seconds = 0.0
setattr(litellm, setting, value)
with pytest.raises(ValueError, match=setting):
PrometheusLogger()
with caplog.at_level(logging.WARNING, logger="LiteLLM"):
logger: Final = PrometheusLogger()
for index in range(3):
_count_request(logger, f"agent-{index}")
clock[0] += 100.0
series: Final = _scraped_series("litellm_proxy_total_requests_metric_total")
assert _label_values(series, "user_agent") == {"agent-0", "agent-1", "agent-2"}
assert setting in caplog.text