From f5de33a4a5e14881605954f4424bc0a40ee6a55d Mon Sep 17 00:00:00 2001 From: Josh Date: Fri, 10 Apr 2026 16:00:08 -0400 Subject: [PATCH] feat(prometheus): reduce default latency bucket cardinality and make configurable --- litellm/__init__.py | 1 + litellm/integrations/prometheus.py | 17 +++++--- litellm/integrations/prometheus_services.py | 8 +++- litellm/types/integrations/prometheus.py | 26 ++----------- .../test_prometheus_user_team_metrics.py | 39 +++++++++++++++++++ 5 files changed, 62 insertions(+), 29 deletions(-) diff --git a/litellm/__init__.py b/litellm/__init__.py index 64c60ca3374..b796c89484a 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -164,6 +164,7 @@ initialized_langfuse_clients: int = 0 langfuse_default_tags: Optional[List[str]] = None langsmith_batch_size: Optional[int] = None prometheus_initialize_budget_metrics: Optional[bool] = False +prometheus_latency_buckets: Optional[List[float]] = None require_auth_for_metrics_endpoint: Optional[bool] = False argilla_batch_size: Optional[int] = None datadog_use_v1: Optional[bool] = False # if you want to use v1 datadog logged payload. diff --git a/litellm/integrations/prometheus.py b/litellm/integrations/prometheus.py index c395987695b..b3bf792e93b 100644 --- a/litellm/integrations/prometheus.py +++ b/litellm/integrations/prometheus.py @@ -86,6 +86,11 @@ class PrometheusLogger(CustomLogger): # Always initialize label_filters, even for non-premium users self.label_filters = self._parse_prometheus_config() + _custom_buckets = litellm.prometheus_latency_buckets + self.latency_buckets = ( + tuple(_custom_buckets) if _custom_buckets is not None else LATENCY_BUCKETS + ) + # Create metric factory functions self._counter_factory = self._create_metric_factory(Counter) self._gauge_factory = self._create_metric_factory(Gauge) @@ -114,14 +119,14 @@ class PrometheusLogger(CustomLogger): labelnames=self.get_labels_for_metric( "litellm_request_total_latency_metric" ), - buckets=LATENCY_BUCKETS, + buckets=self.latency_buckets, ) self.litellm_llm_api_latency_metric = self._histogram_factory( "litellm_llm_api_latency_metric", "Total latency (seconds) for a models LLM API call", labelnames=self.get_labels_for_metric("litellm_llm_api_latency_metric"), - buckets=LATENCY_BUCKETS, + buckets=self.latency_buckets, ) self.litellm_llm_api_time_to_first_token_metric = self._histogram_factory( @@ -137,7 +142,7 @@ class PrometheusLogger(CustomLogger): labelnames=self.get_labels_for_metric( "litellm_llm_api_time_to_first_token_metric" ), - buckets=LATENCY_BUCKETS, + buckets=self.latency_buckets, ) # Counter for spend @@ -314,7 +319,7 @@ class PrometheusLogger(CustomLogger): labelnames=self.get_labels_for_metric( "litellm_overhead_latency_metric" ), - buckets=LATENCY_BUCKETS, + buckets=self.latency_buckets, ) # Request queue time metric @@ -324,7 +329,7 @@ class PrometheusLogger(CustomLogger): labelnames=self.get_labels_for_metric( "litellm_request_queue_time_seconds" ), - buckets=LATENCY_BUCKETS, + buckets=self.latency_buckets, ) # Guardrail metrics @@ -332,7 +337,7 @@ class PrometheusLogger(CustomLogger): "litellm_guardrail_latency_seconds", "Latency (seconds) for guardrail execution", labelnames=["guardrail_name", "status", "error_type", "hook_type"], - buckets=LATENCY_BUCKETS, + buckets=self.latency_buckets, ) self.litellm_guardrail_errors_total = self._counter_factory( diff --git a/litellm/integrations/prometheus_services.py b/litellm/integrations/prometheus_services.py index 55ce758ece6..6d549470613 100644 --- a/litellm/integrations/prometheus_services.py +++ b/litellm/integrations/prometheus_services.py @@ -5,6 +5,7 @@ from typing import Dict, List, Optional, Union +import litellm from litellm._logging import print_verbose, verbose_logger from litellm.types.integrations.prometheus import LATENCY_BUCKETS from litellm.types.services import ( @@ -35,6 +36,11 @@ class PrometheusServicesLogger: "Missing prometheus_client. Run `pip install prometheus-client`" ) + _custom_buckets = litellm.prometheus_latency_buckets + self.latency_buckets = ( + tuple(_custom_buckets) if _custom_buckets is not None else LATENCY_BUCKETS + ) + self.Histogram = Histogram self.Counter = Counter self.Gauge = Gauge @@ -130,7 +136,7 @@ class PrometheusServicesLogger: metric_name, "Latency for {} service".format(service), labelnames=[service], - buckets=LATENCY_BUCKETS, + buckets=self.latency_buckets, ) def create_gauge(self, service: str, type_of_request: str): diff --git a/litellm/types/integrations/prometheus.py b/litellm/types/integrations/prometheus.py index 5f1aa9fb2ce..51a41f97e03 100644 --- a/litellm/types/integrations/prometheus.py +++ b/litellm/types/integrations/prometheus.py @@ -122,40 +122,22 @@ STATUS_CODE = "status_code" EXCEPTION_LABELS = [EXCEPTION_STATUS, EXCEPTION_CLASS] LATENCY_BUCKETS = ( 0.005, - 0.00625, - 0.0125, + 0.01, 0.025, 0.05, 0.1, + 0.25, 0.5, 1.0, - 1.5, 2.0, - 2.5, - 3.0, - 3.5, - 4.0, - 4.5, 5.0, - 5.5, - 6.0, - 6.5, - 7.0, - 7.5, - 8.0, - 8.5, - 9.0, - 9.5, 10.0, - 15.0, - 20.0, - 25.0, 30.0, 60.0, 120.0, - 180.0, - 240.0, 300.0, + 420.0, # 7 minutes + 600.0, # 10 minutes (typical default LLM request timeout) float("inf"), ) diff --git a/tests/test_litellm/integrations/test_prometheus_user_team_metrics.py b/tests/test_litellm/integrations/test_prometheus_user_team_metrics.py index 6b65f444046..e056284ed38 100644 --- a/tests/test_litellm/integrations/test_prometheus_user_team_metrics.py +++ b/tests/test_litellm/integrations/test_prometheus_user_team_metrics.py @@ -768,3 +768,42 @@ async def test_initialize_org_budget_metrics(prometheus_logger): prometheus_logger.litellm_org_max_budget_metric.labels().set.assert_called_once_with( 500.0 ) + + +def test_default_latency_buckets(prometheus_logger): + """PrometheusLogger uses the new reduced default latency buckets.""" + from litellm.types.integrations.prometheus import LATENCY_BUCKETS + + assert prometheus_logger.latency_buckets == LATENCY_BUCKETS + # 420 and 600 should be present + assert 420.0 in prometheus_logger.latency_buckets + assert 600.0 in prometheus_logger.latency_buckets + # dense half-second buckets from old defaults should be gone + assert 1.5 not in prometheus_logger.latency_buckets + assert 9.5 not in prometheus_logger.latency_buckets + + +def test_custom_latency_buckets(): + """prometheus_latency_buckets in litellm settings overrides the defaults.""" + import litellm + from prometheus_client import REGISTRY + + custom_buckets = [0.1, 0.5, 1.0, 5.0, 10.0] + original = litellm.prometheus_latency_buckets + # Clear registry before creating a new PrometheusLogger + for collector in list(REGISTRY._collector_to_names.keys()): + try: + REGISTRY.unregister(collector) + except Exception: + pass + try: + litellm.prometheus_latency_buckets = custom_buckets + logger = PrometheusLogger() + assert logger.latency_buckets == tuple(custom_buckets) + finally: + litellm.prometheus_latency_buckets = original + for collector in list(REGISTRY._collector_to_names.keys()): + try: + REGISTRY.unregister(collector) + except Exception: + pass