feat(prometheus): reduce default latency bucket cardinality and make configurable

This commit is contained in:
Josh 2026-04-10 16:00:08 -04:00
parent d0e347af32
commit f5de33a4a5
5 changed files with 62 additions and 29 deletions

View file

@ -164,6 +164,7 @@ initialized_langfuse_clients: int = 0
langfuse_default_tags: Optional[List[str]] = None
langsmith_batch_size: Optional[int] = None
prometheus_initialize_budget_metrics: Optional[bool] = False
prometheus_latency_buckets: Optional[List[float]] = None
require_auth_for_metrics_endpoint: Optional[bool] = False
argilla_batch_size: Optional[int] = None
datadog_use_v1: Optional[bool] = False # if you want to use v1 datadog logged payload.

View file

@ -86,6 +86,11 @@ class PrometheusLogger(CustomLogger):
# Always initialize label_filters, even for non-premium users
self.label_filters = self._parse_prometheus_config()
_custom_buckets = litellm.prometheus_latency_buckets
self.latency_buckets = (
tuple(_custom_buckets) if _custom_buckets is not None else LATENCY_BUCKETS
)
# Create metric factory functions
self._counter_factory = self._create_metric_factory(Counter)
self._gauge_factory = self._create_metric_factory(Gauge)
@ -114,14 +119,14 @@ class PrometheusLogger(CustomLogger):
labelnames=self.get_labels_for_metric(
"litellm_request_total_latency_metric"
),
buckets=LATENCY_BUCKETS,
buckets=self.latency_buckets,
)
self.litellm_llm_api_latency_metric = self._histogram_factory(
"litellm_llm_api_latency_metric",
"Total latency (seconds) for a models LLM API call",
labelnames=self.get_labels_for_metric("litellm_llm_api_latency_metric"),
buckets=LATENCY_BUCKETS,
buckets=self.latency_buckets,
)
self.litellm_llm_api_time_to_first_token_metric = self._histogram_factory(
@ -137,7 +142,7 @@ class PrometheusLogger(CustomLogger):
labelnames=self.get_labels_for_metric(
"litellm_llm_api_time_to_first_token_metric"
),
buckets=LATENCY_BUCKETS,
buckets=self.latency_buckets,
)
# Counter for spend
@ -314,7 +319,7 @@ class PrometheusLogger(CustomLogger):
labelnames=self.get_labels_for_metric(
"litellm_overhead_latency_metric"
),
buckets=LATENCY_BUCKETS,
buckets=self.latency_buckets,
)
# Request queue time metric
@ -324,7 +329,7 @@ class PrometheusLogger(CustomLogger):
labelnames=self.get_labels_for_metric(
"litellm_request_queue_time_seconds"
),
buckets=LATENCY_BUCKETS,
buckets=self.latency_buckets,
)
# Guardrail metrics
@ -332,7 +337,7 @@ class PrometheusLogger(CustomLogger):
"litellm_guardrail_latency_seconds",
"Latency (seconds) for guardrail execution",
labelnames=["guardrail_name", "status", "error_type", "hook_type"],
buckets=LATENCY_BUCKETS,
buckets=self.latency_buckets,
)
self.litellm_guardrail_errors_total = self._counter_factory(

View file

@ -5,6 +5,7 @@
from typing import Dict, List, Optional, Union
import litellm
from litellm._logging import print_verbose, verbose_logger
from litellm.types.integrations.prometheus import LATENCY_BUCKETS
from litellm.types.services import (
@ -35,6 +36,11 @@ class PrometheusServicesLogger:
"Missing prometheus_client. Run `pip install prometheus-client`"
)
_custom_buckets = litellm.prometheus_latency_buckets
self.latency_buckets = (
tuple(_custom_buckets) if _custom_buckets is not None else LATENCY_BUCKETS
)
self.Histogram = Histogram
self.Counter = Counter
self.Gauge = Gauge
@ -130,7 +136,7 @@ class PrometheusServicesLogger:
metric_name,
"Latency for {} service".format(service),
labelnames=[service],
buckets=LATENCY_BUCKETS,
buckets=self.latency_buckets,
)
def create_gauge(self, service: str, type_of_request: str):

View file

@ -122,40 +122,22 @@ STATUS_CODE = "status_code"
EXCEPTION_LABELS = [EXCEPTION_STATUS, EXCEPTION_CLASS]
LATENCY_BUCKETS = (
0.005,
0.00625,
0.0125,
0.01,
0.025,
0.05,
0.1,
0.25,
0.5,
1.0,
1.5,
2.0,
2.5,
3.0,
3.5,
4.0,
4.5,
5.0,
5.5,
6.0,
6.5,
7.0,
7.5,
8.0,
8.5,
9.0,
9.5,
10.0,
15.0,
20.0,
25.0,
30.0,
60.0,
120.0,
180.0,
240.0,
300.0,
420.0, # 7 minutes
600.0, # 10 minutes (typical default LLM request timeout)
float("inf"),
)

View file

@ -768,3 +768,42 @@ async def test_initialize_org_budget_metrics(prometheus_logger):
prometheus_logger.litellm_org_max_budget_metric.labels().set.assert_called_once_with(
500.0
)
def test_default_latency_buckets(prometheus_logger):
"""PrometheusLogger uses the new reduced default latency buckets."""
from litellm.types.integrations.prometheus import LATENCY_BUCKETS
assert prometheus_logger.latency_buckets == LATENCY_BUCKETS
# 420 and 600 should be present
assert 420.0 in prometheus_logger.latency_buckets
assert 600.0 in prometheus_logger.latency_buckets
# dense half-second buckets from old defaults should be gone
assert 1.5 not in prometheus_logger.latency_buckets
assert 9.5 not in prometheus_logger.latency_buckets
def test_custom_latency_buckets():
"""prometheus_latency_buckets in litellm settings overrides the defaults."""
import litellm
from prometheus_client import REGISTRY
custom_buckets = [0.1, 0.5, 1.0, 5.0, 10.0]
original = litellm.prometheus_latency_buckets
# Clear registry before creating a new PrometheusLogger
for collector in list(REGISTRY._collector_to_names.keys()):
try:
REGISTRY.unregister(collector)
except Exception:
pass
try:
litellm.prometheus_latency_buckets = custom_buckets
logger = PrometheusLogger()
assert logger.latency_buckets == tuple(custom_buckets)
finally:
litellm.prometheus_latency_buckets = original
for collector in list(REGISTRY._collector_to_names.keys()):
try:
REGISTRY.unregister(collector)
except Exception:
pass