feat(prometheus): configurable latency buckets, exclude_metrics, exclude_labels

This commit is contained in:
Ishaan Jaffer 2026-03-23 11:47:44 -07:00
parent 742e176611
commit 1270ef7be0
3 changed files with 191 additions and 86 deletions

View file

@ -1,3 +1,60 @@
# Prometheus Metrics Configuration
## Custom Latency Buckets
By default, LiteLLM uses a fixed set of histogram buckets for all latency metrics. You can override them with your own tuple of bucket boundaries.
```python
import litellm
litellm.prometheus_latency_buckets = (0.05, 0.1, 0.25, 0.5, 1.0, 2.0, 5.0, float("inf"))
```
This applies to all histogram metrics:
- `litellm_request_total_latency_metric`
- `litellm_llm_api_latency_metric`
- `litellm_llm_api_time_to_first_token_metric`
- `litellm_overhead_latency_metric`
- `litellm_request_queue_time_seconds`
- `litellm_guardrail_latency_seconds`
> Set this **before** the Prometheus logger is initialized (i.e. before the proxy starts).
---
## Exclude Metrics
Disable specific metrics entirely so they are never registered or emitted.
```python
import litellm
litellm.prometheus_exclude_metrics = [
"litellm_overhead_latency_metric",
"litellm_request_queue_time_seconds",
]
```
Any metric in this list is replaced by a no-op — no Prometheus series will be created for it.
---
## Exclude Labels
Strip specific label dimensions from **all** metrics. Useful for reducing cardinality.
```python
import litellm
litellm.prometheus_exclude_labels = ["end_user", "user_agent", "client_ip"]
```
The label will be removed from every metric that would normally carry it.
> `prometheus_exclude_labels` is applied **after** any `prometheus_metrics_config` include-label filtering.
---
# 💸 GET Daily Spend, Usage Metrics
## Request Format

View file

@ -167,12 +167,12 @@ prometheus_initialize_budget_metrics: Optional[bool] = False
require_auth_for_metrics_endpoint: Optional[bool] = False
argilla_batch_size: Optional[int] = None
datadog_use_v1: Optional[bool] = False # if you want to use v1 datadog logged payload.
gcs_pub_sub_use_v1: Optional[
bool
] = False # if you want to use v1 gcs pubsub logged payload
generic_api_use_v1: Optional[
bool
] = False # if you want to use v1 generic api logged payload
gcs_pub_sub_use_v1: Optional[bool] = (
False # if you want to use v1 gcs pubsub logged payload
)
generic_api_use_v1: Optional[bool] = (
False # if you want to use v1 generic api logged payload
)
argilla_transformation_object: Optional[Dict[str, Any]] = None
_async_input_callback: List[
Union[str, Callable, "CustomLogger"]
@ -192,25 +192,25 @@ _async_failure_callback: List[
pre_call_rules: List[Callable] = []
post_call_rules: List[Callable] = []
turn_off_message_logging: Optional[bool] = False
standard_logging_payload_excluded_fields: Optional[
List[str]
] = None # Fields to exclude from StandardLoggingPayload before callbacks receive it
standard_logging_payload_excluded_fields: Optional[List[str]] = (
None # Fields to exclude from StandardLoggingPayload before callbacks receive it
)
log_raw_request_response: bool = False
redact_messages_in_exceptions: Optional[bool] = False
redact_user_api_key_info: Optional[bool] = False
filter_invalid_headers: Optional[bool] = False
add_user_information_to_llm_headers: Optional[
bool
] = None # adds user_id, team_id, token hash (params from StandardLoggingMetadata) to request headers
add_user_information_to_llm_headers: Optional[bool] = (
None # adds user_id, team_id, token hash (params from StandardLoggingMetadata) to request headers
)
store_audit_logs = False # Enterprise feature, allow users to see audit logs
### end of callbacks #############
email: Optional[
str
] = None # Not used anymore, will be removed in next MAJOR release - https://github.com/BerriAI/litellm/discussions/648
token: Optional[
str
] = None # Not used anymore, will be removed in next MAJOR release - https://github.com/BerriAI/litellm/discussions/648
email: Optional[str] = (
None # Not used anymore, will be removed in next MAJOR release - https://github.com/BerriAI/litellm/discussions/648
)
token: Optional[str] = (
None # Not used anymore, will be removed in next MAJOR release - https://github.com/BerriAI/litellm/discussions/648
)
telemetry = True
max_tokens: int = DEFAULT_MAX_TOKENS # OpenAI Defaults
drop_params = bool(os.getenv("LITELLM_DROP_PARAMS", False))
@ -272,9 +272,9 @@ use_client: bool = False
ssl_verify: Union[str, bool] = True
ssl_security_level: Optional[str] = None
ssl_certificate: Optional[str] = None
ssl_ecdh_curve: Optional[
str
] = None # Set to 'X25519' to disable PQC and improve performance
ssl_ecdh_curve: Optional[str] = (
None # Set to 'X25519' to disable PQC and improve performance
)
disable_streaming_logging: bool = False
disable_token_counter: bool = False
disable_add_transform_inline_image_block: bool = False
@ -327,20 +327,24 @@ enable_loadbalancing_on_batch_endpoints: Optional[bool] = None
enable_caching_on_provider_specific_optional_params: bool = (
False # feature-flag for caching on optional params - e.g. 'top_k'
)
caching: bool = False # Not used anymore, will be removed in next MAJOR release - https://github.com/BerriAI/litellm/discussions/648
caching_with_models: bool = False # # Not used anymore, will be removed in next MAJOR release - https://github.com/BerriAI/litellm/discussions/648
cache: Optional[
"Cache"
] = None # cache object <- use this - https://docs.litellm.ai/docs/caching
caching: bool = (
False # Not used anymore, will be removed in next MAJOR release - https://github.com/BerriAI/litellm/discussions/648
)
caching_with_models: bool = (
False # # Not used anymore, will be removed in next MAJOR release - https://github.com/BerriAI/litellm/discussions/648
)
cache: Optional["Cache"] = (
None # cache object <- use this - https://docs.litellm.ai/docs/caching
)
default_in_memory_ttl: Optional[float] = None
default_redis_ttl: Optional[float] = None
default_redis_batch_cache_expiry: Optional[float] = None
model_alias_map: Dict[str, str] = {}
model_group_settings: Optional["ModelGroupSettings"] = None
max_budget: float = 0.0 # set the max budget across all providers
budget_duration: Optional[
str
] = None # proxy only - resets budget after fixed duration. You can set duration as seconds ("30s"), minutes ("30m"), hours ("30h"), days ("30d").
budget_duration: Optional[str] = (
None # proxy only - resets budget after fixed duration. You can set duration as seconds ("30s"), minutes ("30m"), hours ("30h"), days ("30d").
)
default_soft_budget: float = (
DEFAULT_SOFT_BUDGET # by default all litellm proxy keys have a soft budget of 50.0
)
@ -349,7 +353,9 @@ forward_traceparent_to_llm_provider: bool = False
_current_cost = 0.0 # private variable, used if max budget is set
error_logs: Dict = {}
add_function_to_prompt: bool = False # if function calling not supported by api, append function call details to system prompt
add_function_to_prompt: bool = (
False # if function calling not supported by api, append function call details to system prompt
)
client_session: Optional[httpx.Client] = None
aclient_session: Optional[httpx.AsyncClient] = None
model_fallbacks: Optional[List] = None # Deprecated for 'litellm.fallbacks'
@ -392,11 +398,22 @@ enable_end_user_cost_tracking_prometheus_only: Optional[bool] = None
custom_prometheus_metadata_labels: List[str] = []
custom_prometheus_tags: List[str] = []
prometheus_metrics_config: Optional[List] = None
prometheus_latency_buckets: Optional[Tuple] = (
None # override default LATENCY_BUCKETS for histogram metrics
)
prometheus_exclude_metrics: Optional[List[str]] = (
None # metric names to disable entirely
)
prometheus_exclude_labels: Optional[List[str]] = (
None # label names to strip from all metrics
)
prometheus_emit_stream_label: bool = False
disable_add_prefix_to_prompt: bool = (
False # used by anthropic, to disable adding prefix to prompt
)
disable_copilot_system_to_assistant: bool = False # If false (default), converts all 'system' role messages to 'assistant' for GitHub Copilot compatibility. Set to true to disable this behavior.
disable_copilot_system_to_assistant: bool = (
False # If false (default), converts all 'system' role messages to 'assistant' for GitHub Copilot compatibility. Set to true to disable this behavior.
)
public_mcp_servers: Optional[List[str]] = None
public_model_groups: Optional[List[str]] = None
public_agent_groups: Optional[List[str]] = None
@ -405,9 +422,9 @@ public_agent_groups: Optional[List[str]] = None
# Old format: { "displayName": "url" } (for backward compatibility)
public_model_groups_links: Dict[str, Union[str, Dict[str, Any]]] = {}
#### REQUEST PRIORITIZATION #######
priority_reservation: Optional[
Dict[str, Union[float, "PriorityReservationDict"]]
] = None
priority_reservation: Optional[Dict[str, Union[float, "PriorityReservationDict"]]] = (
None
)
# priority_reservation_settings is lazy-loaded via __getattr__
# Only declare for type checking - at runtime __getattr__ handles it
if TYPE_CHECKING:
@ -415,13 +432,17 @@ if TYPE_CHECKING:
######## Networking Settings ########
use_aiohttp_transport: bool = True # Older variable, aiohttp is now the default. use disable_aiohttp_transport instead.
use_aiohttp_transport: bool = (
True # Older variable, aiohttp is now the default. use disable_aiohttp_transport instead.
)
aiohttp_trust_env: bool = False # set to true to use HTTP_ Proxy settings
disable_aiohttp_transport: bool = False # Set this to true to use httpx instead
disable_aiohttp_trust_env: bool = (
False # When False, aiohttp will respect HTTP(S)_PROXY env vars
)
force_ipv4: bool = False # when True, litellm will force ipv4 for all LLM requests. Some users have seen httpx ConnectionError when using ipv6.
force_ipv4: bool = (
False # when True, litellm will force ipv4 for all LLM requests. Some users have seen httpx ConnectionError when using ipv6.
)
network_mock: bool = False # When True, use mock transport — no real network calls
####### STOP SEQUENCE LIMIT #######
@ -436,13 +457,13 @@ context_window_fallbacks: Optional[List] = None
content_policy_fallbacks: Optional[List] = None
allowed_fails: int = 3
allow_dynamic_callback_disabling: bool = True
num_retries_per_request: Optional[
int
] = None # for the request overall (incl. fallbacks + model retries)
num_retries_per_request: Optional[int] = (
None # for the request overall (incl. fallbacks + model retries)
)
####### SECRET MANAGERS #####################
secret_manager_client: Optional[
Any
] = None # list of instantiated key management clients - e.g. azure kv, infisical, etc.
secret_manager_client: Optional[Any] = (
None # list of instantiated key management clients - e.g. azure kv, infisical, etc.
)
_google_kms_resource_name: Optional[str] = None
_key_management_system: Optional["KeyManagementSystem"] = None
# Note: KeyManagementSettings must be eagerly imported because _key_management_settings
@ -455,12 +476,12 @@ output_parse_pii: bool = False
from litellm.litellm_core_utils.get_model_cost_map import get_model_cost_map
model_cost = get_model_cost_map(url=model_cost_map_url)
cost_discount_config: Dict[
str, float
] = {} # Provider-specific cost discounts {"vertex_ai": 0.05} = 5% discount
cost_margin_config: Dict[
str, Union[float, Dict[str, float]]
] = {} # Provider-specific or global cost margins. Examples:
cost_discount_config: Dict[str, float] = (
{}
) # Provider-specific cost discounts {"vertex_ai": 0.05} = 5% discount
cost_margin_config: Dict[str, Union[float, Dict[str, float]]] = (
{}
) # Provider-specific or global cost margins. Examples:
# Percentage: {"openai": 0.10} = 10% margin
# Fixed: {"openai": {"fixed_amount": 0.001}} = $0.001 per request
# Global: {"global": 0.05} = 5% global margin on all providers
@ -1309,12 +1330,12 @@ from . import rag
from .types.llms.custom_llm import CustomLLMItem
custom_provider_map: List[CustomLLMItem] = []
_custom_providers: List[
str
] = [] # internal helper util, used to track names of custom providers
disable_hf_tokenizer_download: Optional[
bool
] = None # disable huggingface tokenizer download. Defaults to openai clk100
_custom_providers: List[str] = (
[]
) # internal helper util, used to track names of custom providers
disable_hf_tokenizer_download: Optional[bool] = (
None # disable huggingface tokenizer download. Defaults to openai clk100
)
global_disable_no_log_param: bool = False
### CLI UTILITIES ###

View file

@ -74,6 +74,7 @@ class PrometheusLogger(CustomLogger):
# Always initialize label_filters, even for non-premium users
self.label_filters = self._parse_prometheus_config()
self._parse_exclude_config()
# Create metric factory functions
self._counter_factory = self._create_metric_factory(Counter)
@ -103,14 +104,14 @@ class PrometheusLogger(CustomLogger):
labelnames=self.get_labels_for_metric(
"litellm_request_total_latency_metric"
),
buckets=LATENCY_BUCKETS,
buckets=self._get_latency_buckets(),
)
self.litellm_llm_api_latency_metric = self._histogram_factory(
"litellm_llm_api_latency_metric",
"Total latency (seconds) for a models LLM API call",
labelnames=self.get_labels_for_metric("litellm_llm_api_latency_metric"),
buckets=LATENCY_BUCKETS,
buckets=self._get_latency_buckets(),
)
self.litellm_llm_api_time_to_first_token_metric = self._histogram_factory(
@ -126,7 +127,7 @@ class PrometheusLogger(CustomLogger):
labelnames=self.get_labels_for_metric(
"litellm_llm_api_time_to_first_token_metric"
),
buckets=LATENCY_BUCKETS,
buckets=self._get_latency_buckets(),
)
# Counter for spend
@ -278,7 +279,7 @@ class PrometheusLogger(CustomLogger):
labelnames=self.get_labels_for_metric(
"litellm_overhead_latency_metric"
),
buckets=LATENCY_BUCKETS,
buckets=self._get_latency_buckets(),
)
# Request queue time metric
@ -288,7 +289,7 @@ class PrometheusLogger(CustomLogger):
labelnames=self.get_labels_for_metric(
"litellm_request_queue_time_seconds"
),
buckets=LATENCY_BUCKETS,
buckets=self._get_latency_buckets(),
)
# Guardrail metrics
@ -296,7 +297,7 @@ class PrometheusLogger(CustomLogger):
"litellm_guardrail_latency_seconds",
"Latency (seconds) for guardrail execution",
labelnames=["guardrail_name", "status", "error_type", "hook_type"],
buckets=LATENCY_BUCKETS,
buckets=self._get_latency_buckets(),
)
self.litellm_guardrail_errors_total = self._counter_factory(
@ -444,6 +445,26 @@ class PrometheusLogger(CustomLogger):
print_verbose(f"Got exception on init prometheus client {str(e)}")
raise e
def _get_latency_buckets(self) -> tuple:
"""Return latency buckets to use for histogram metrics.
Uses ``litellm.prometheus_latency_buckets`` when set, falling back to the
module-level ``LATENCY_BUCKETS`` constant.
"""
import litellm
return litellm.prometheus_latency_buckets or LATENCY_BUCKETS
def _parse_exclude_config(self) -> None:
"""Populate self.excluded_metrics and self.excluded_labels from litellm module vars."""
import litellm
raw_metrics = litellm.prometheus_exclude_metrics or []
raw_labels = litellm.prometheus_exclude_labels or []
self.excluded_metrics: set = set(raw_metrics)
self.excluded_labels: set = set(raw_labels)
def _parse_prometheus_config(self) -> Dict[str, List[str]]:
"""Parse prometheus metrics configuration for label filtering and enabled metrics"""
import litellm
@ -835,7 +856,11 @@ class PrometheusLogger(CustomLogger):
def _is_metric_enabled(self, metric_name: str) -> bool:
"""Check if a metric is enabled based on configuration"""
# If no specific configuration is provided, enable all metrics (default behavior)
# Check exclude list first — excluded metrics are always disabled
if hasattr(self, "excluded_metrics") and metric_name in self.excluded_metrics:
return False
# If no specific include configuration is provided, enable all metrics (default behavior)
if not hasattr(self, "enabled_metrics"):
return True
@ -870,17 +895,17 @@ class PrometheusLogger(CustomLogger):
# If no label filtering is configured for this metric, use default labels
if metric_name not in self.label_filters:
return default_labels
labels = default_labels
else:
# Return intersection of configured and default labels to ensure we only use valid labels
configured_labels = self.label_filters[metric_name]
labels = [label for label in default_labels if label in configured_labels]
# Get configured labels for this metric
configured_labels = self.label_filters[metric_name]
# Strip globally excluded labels
if hasattr(self, "excluded_labels") and self.excluded_labels:
labels = [label for label in labels if label not in self.excluded_labels]
# Return intersection of configured and default labels to ensure we only use valid labels
filtered_labels = [
label for label in default_labels if label in configured_labels
]
return filtered_labels
return labels
async def async_log_success_event(self, kwargs, response_obj, start_time, end_time):
# Define prometheus client
@ -978,9 +1003,11 @@ class PrometheusLogger(CustomLogger):
),
client_ip=standard_logging_payload["metadata"].get("requester_ip_address"),
user_agent=standard_logging_payload["metadata"].get("user_agent"),
stream=str(standard_logging_payload.get("stream"))
if litellm.prometheus_emit_stream_label
else None,
stream=(
str(standard_logging_payload.get("stream"))
if litellm.prometheus_emit_stream_label
else None
),
)
if (
@ -1633,9 +1660,11 @@ class PrometheusLogger(CustomLogger):
client_ip=_metadata.get("requester_ip_address"),
user_agent=_metadata.get("user_agent"),
model_id=model_id,
stream=str(request_data.get("stream"))
if litellm.prometheus_emit_stream_label
else None,
stream=(
str(request_data.get("stream"))
if litellm.prometheus_emit_stream_label
else None
),
)
_labels = prometheus_label_factory(
supported_enum_labels=self.get_labels_for_metric(
@ -1959,9 +1988,9 @@ class PrometheusLogger(CustomLogger):
):
try:
verbose_logger.debug("setting remaining tokens requests metric")
standard_logging_payload: Optional[
StandardLoggingPayload
] = request_kwargs.get("standard_logging_object")
standard_logging_payload: Optional[StandardLoggingPayload] = (
request_kwargs.get("standard_logging_object")
)
if standard_logging_payload is None:
return
@ -2473,9 +2502,7 @@ class PrometheusLogger(CustomLogger):
)
return
async def fetch_keys(
page_size: int, page: int
) -> Tuple[
async def fetch_keys(page_size: int, page: int) -> Tuple[
List[Union[str, UserAPIKeyAuth, LiteLLM_DeletedVerificationToken]],
Optional[int],
]:
@ -2999,10 +3026,10 @@ class PrometheusLogger(CustomLogger):
from litellm.constants import PROMETHEUS_BUDGET_METRICS_REFRESH_INTERVAL_MINUTES
from litellm.integrations.custom_logger import CustomLogger
prometheus_loggers: List[
CustomLogger
] = litellm.logging_callback_manager.get_custom_loggers_for_type(
callback_type=PrometheusLogger
prometheus_loggers: List[CustomLogger] = (
litellm.logging_callback_manager.get_custom_loggers_for_type(
callback_type=PrometheusLogger
)
)
# we need to get the initialized prometheus logger instance(s) and call logger.initialize_remaining_budget_metrics() on them
verbose_logger.debug("found %s prometheus loggers", len(prometheus_loggers))