From 1270ef7be084cc1c1923d314eb6c8af85a02bdeb Mon Sep 17 00:00:00 2001 From: Ishaan Jaffer Date: Mon, 23 Mar 2026 11:47:44 -0700 Subject: [PATCH] feat(prometheus): configurable latency buckets, exclude_metrics, exclude_labels --- docs/my-website/docs/proxy/metrics.md | 57 ++++++++++++ litellm/__init__.py | 129 +++++++++++++++----------- litellm/integrations/prometheus.py | 91 +++++++++++------- 3 files changed, 191 insertions(+), 86 deletions(-) diff --git a/docs/my-website/docs/proxy/metrics.md b/docs/my-website/docs/proxy/metrics.md index bf5ebe2858e..891b27f38ed 100644 --- a/docs/my-website/docs/proxy/metrics.md +++ b/docs/my-website/docs/proxy/metrics.md @@ -1,3 +1,60 @@ +# Prometheus Metrics Configuration + +## Custom Latency Buckets + +By default, LiteLLM uses a fixed set of histogram buckets for all latency metrics. You can override them with your own tuple of bucket boundaries. + +```python +import litellm + +litellm.prometheus_latency_buckets = (0.05, 0.1, 0.25, 0.5, 1.0, 2.0, 5.0, float("inf")) +``` + +This applies to all histogram metrics: +- `litellm_request_total_latency_metric` +- `litellm_llm_api_latency_metric` +- `litellm_llm_api_time_to_first_token_metric` +- `litellm_overhead_latency_metric` +- `litellm_request_queue_time_seconds` +- `litellm_guardrail_latency_seconds` + +> Set this **before** the Prometheus logger is initialized (i.e. before the proxy starts). + +--- + +## Exclude Metrics + +Disable specific metrics entirely so they are never registered or emitted. + +```python +import litellm + +litellm.prometheus_exclude_metrics = [ + "litellm_overhead_latency_metric", + "litellm_request_queue_time_seconds", +] +``` + +Any metric in this list is replaced by a no-op — no Prometheus series will be created for it. + +--- + +## Exclude Labels + +Strip specific label dimensions from **all** metrics. Useful for reducing cardinality. + +```python +import litellm + +litellm.prometheus_exclude_labels = ["end_user", "user_agent", "client_ip"] +``` + +The label will be removed from every metric that would normally carry it. + +> `prometheus_exclude_labels` is applied **after** any `prometheus_metrics_config` include-label filtering. + +--- + # 💸 GET Daily Spend, Usage Metrics ## Request Format diff --git a/litellm/__init__.py b/litellm/__init__.py index e45d926e8db..72dd135b386 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -167,12 +167,12 @@ prometheus_initialize_budget_metrics: Optional[bool] = False require_auth_for_metrics_endpoint: Optional[bool] = False argilla_batch_size: Optional[int] = None datadog_use_v1: Optional[bool] = False # if you want to use v1 datadog logged payload. -gcs_pub_sub_use_v1: Optional[ - bool -] = False # if you want to use v1 gcs pubsub logged payload -generic_api_use_v1: Optional[ - bool -] = False # if you want to use v1 generic api logged payload +gcs_pub_sub_use_v1: Optional[bool] = ( + False # if you want to use v1 gcs pubsub logged payload +) +generic_api_use_v1: Optional[bool] = ( + False # if you want to use v1 generic api logged payload +) argilla_transformation_object: Optional[Dict[str, Any]] = None _async_input_callback: List[ Union[str, Callable, "CustomLogger"] @@ -192,25 +192,25 @@ _async_failure_callback: List[ pre_call_rules: List[Callable] = [] post_call_rules: List[Callable] = [] turn_off_message_logging: Optional[bool] = False -standard_logging_payload_excluded_fields: Optional[ - List[str] -] = None # Fields to exclude from StandardLoggingPayload before callbacks receive it +standard_logging_payload_excluded_fields: Optional[List[str]] = ( + None # Fields to exclude from StandardLoggingPayload before callbacks receive it +) log_raw_request_response: bool = False redact_messages_in_exceptions: Optional[bool] = False redact_user_api_key_info: Optional[bool] = False filter_invalid_headers: Optional[bool] = False -add_user_information_to_llm_headers: Optional[ - bool -] = None # adds user_id, team_id, token hash (params from StandardLoggingMetadata) to request headers +add_user_information_to_llm_headers: Optional[bool] = ( + None # adds user_id, team_id, token hash (params from StandardLoggingMetadata) to request headers +) store_audit_logs = False # Enterprise feature, allow users to see audit logs ### end of callbacks ############# -email: Optional[ - str -] = None # Not used anymore, will be removed in next MAJOR release - https://github.com/BerriAI/litellm/discussions/648 -token: Optional[ - str -] = None # Not used anymore, will be removed in next MAJOR release - https://github.com/BerriAI/litellm/discussions/648 +email: Optional[str] = ( + None # Not used anymore, will be removed in next MAJOR release - https://github.com/BerriAI/litellm/discussions/648 +) +token: Optional[str] = ( + None # Not used anymore, will be removed in next MAJOR release - https://github.com/BerriAI/litellm/discussions/648 +) telemetry = True max_tokens: int = DEFAULT_MAX_TOKENS # OpenAI Defaults drop_params = bool(os.getenv("LITELLM_DROP_PARAMS", False)) @@ -272,9 +272,9 @@ use_client: bool = False ssl_verify: Union[str, bool] = True ssl_security_level: Optional[str] = None ssl_certificate: Optional[str] = None -ssl_ecdh_curve: Optional[ - str -] = None # Set to 'X25519' to disable PQC and improve performance +ssl_ecdh_curve: Optional[str] = ( + None # Set to 'X25519' to disable PQC and improve performance +) disable_streaming_logging: bool = False disable_token_counter: bool = False disable_add_transform_inline_image_block: bool = False @@ -327,20 +327,24 @@ enable_loadbalancing_on_batch_endpoints: Optional[bool] = None enable_caching_on_provider_specific_optional_params: bool = ( False # feature-flag for caching on optional params - e.g. 'top_k' ) -caching: bool = False # Not used anymore, will be removed in next MAJOR release - https://github.com/BerriAI/litellm/discussions/648 -caching_with_models: bool = False # # Not used anymore, will be removed in next MAJOR release - https://github.com/BerriAI/litellm/discussions/648 -cache: Optional[ - "Cache" -] = None # cache object <- use this - https://docs.litellm.ai/docs/caching +caching: bool = ( + False # Not used anymore, will be removed in next MAJOR release - https://github.com/BerriAI/litellm/discussions/648 +) +caching_with_models: bool = ( + False # # Not used anymore, will be removed in next MAJOR release - https://github.com/BerriAI/litellm/discussions/648 +) +cache: Optional["Cache"] = ( + None # cache object <- use this - https://docs.litellm.ai/docs/caching +) default_in_memory_ttl: Optional[float] = None default_redis_ttl: Optional[float] = None default_redis_batch_cache_expiry: Optional[float] = None model_alias_map: Dict[str, str] = {} model_group_settings: Optional["ModelGroupSettings"] = None max_budget: float = 0.0 # set the max budget across all providers -budget_duration: Optional[ - str -] = None # proxy only - resets budget after fixed duration. You can set duration as seconds ("30s"), minutes ("30m"), hours ("30h"), days ("30d"). +budget_duration: Optional[str] = ( + None # proxy only - resets budget after fixed duration. You can set duration as seconds ("30s"), minutes ("30m"), hours ("30h"), days ("30d"). +) default_soft_budget: float = ( DEFAULT_SOFT_BUDGET # by default all litellm proxy keys have a soft budget of 50.0 ) @@ -349,7 +353,9 @@ forward_traceparent_to_llm_provider: bool = False _current_cost = 0.0 # private variable, used if max budget is set error_logs: Dict = {} -add_function_to_prompt: bool = False # if function calling not supported by api, append function call details to system prompt +add_function_to_prompt: bool = ( + False # if function calling not supported by api, append function call details to system prompt +) client_session: Optional[httpx.Client] = None aclient_session: Optional[httpx.AsyncClient] = None model_fallbacks: Optional[List] = None # Deprecated for 'litellm.fallbacks' @@ -392,11 +398,22 @@ enable_end_user_cost_tracking_prometheus_only: Optional[bool] = None custom_prometheus_metadata_labels: List[str] = [] custom_prometheus_tags: List[str] = [] prometheus_metrics_config: Optional[List] = None +prometheus_latency_buckets: Optional[Tuple] = ( + None # override default LATENCY_BUCKETS for histogram metrics +) +prometheus_exclude_metrics: Optional[List[str]] = ( + None # metric names to disable entirely +) +prometheus_exclude_labels: Optional[List[str]] = ( + None # label names to strip from all metrics +) prometheus_emit_stream_label: bool = False disable_add_prefix_to_prompt: bool = ( False # used by anthropic, to disable adding prefix to prompt ) -disable_copilot_system_to_assistant: bool = False # If false (default), converts all 'system' role messages to 'assistant' for GitHub Copilot compatibility. Set to true to disable this behavior. +disable_copilot_system_to_assistant: bool = ( + False # If false (default), converts all 'system' role messages to 'assistant' for GitHub Copilot compatibility. Set to true to disable this behavior. +) public_mcp_servers: Optional[List[str]] = None public_model_groups: Optional[List[str]] = None public_agent_groups: Optional[List[str]] = None @@ -405,9 +422,9 @@ public_agent_groups: Optional[List[str]] = None # Old format: { "displayName": "url" } (for backward compatibility) public_model_groups_links: Dict[str, Union[str, Dict[str, Any]]] = {} #### REQUEST PRIORITIZATION ####### -priority_reservation: Optional[ - Dict[str, Union[float, "PriorityReservationDict"]] -] = None +priority_reservation: Optional[Dict[str, Union[float, "PriorityReservationDict"]]] = ( + None +) # priority_reservation_settings is lazy-loaded via __getattr__ # Only declare for type checking - at runtime __getattr__ handles it if TYPE_CHECKING: @@ -415,13 +432,17 @@ if TYPE_CHECKING: ######## Networking Settings ######## -use_aiohttp_transport: bool = True # Older variable, aiohttp is now the default. use disable_aiohttp_transport instead. +use_aiohttp_transport: bool = ( + True # Older variable, aiohttp is now the default. use disable_aiohttp_transport instead. +) aiohttp_trust_env: bool = False # set to true to use HTTP_ Proxy settings disable_aiohttp_transport: bool = False # Set this to true to use httpx instead disable_aiohttp_trust_env: bool = ( False # When False, aiohttp will respect HTTP(S)_PROXY env vars ) -force_ipv4: bool = False # when True, litellm will force ipv4 for all LLM requests. Some users have seen httpx ConnectionError when using ipv6. +force_ipv4: bool = ( + False # when True, litellm will force ipv4 for all LLM requests. Some users have seen httpx ConnectionError when using ipv6. +) network_mock: bool = False # When True, use mock transport — no real network calls ####### STOP SEQUENCE LIMIT ####### @@ -436,13 +457,13 @@ context_window_fallbacks: Optional[List] = None content_policy_fallbacks: Optional[List] = None allowed_fails: int = 3 allow_dynamic_callback_disabling: bool = True -num_retries_per_request: Optional[ - int -] = None # for the request overall (incl. fallbacks + model retries) +num_retries_per_request: Optional[int] = ( + None # for the request overall (incl. fallbacks + model retries) +) ####### SECRET MANAGERS ##################### -secret_manager_client: Optional[ - Any -] = None # list of instantiated key management clients - e.g. azure kv, infisical, etc. +secret_manager_client: Optional[Any] = ( + None # list of instantiated key management clients - e.g. azure kv, infisical, etc. +) _google_kms_resource_name: Optional[str] = None _key_management_system: Optional["KeyManagementSystem"] = None # Note: KeyManagementSettings must be eagerly imported because _key_management_settings @@ -455,12 +476,12 @@ output_parse_pii: bool = False from litellm.litellm_core_utils.get_model_cost_map import get_model_cost_map model_cost = get_model_cost_map(url=model_cost_map_url) -cost_discount_config: Dict[ - str, float -] = {} # Provider-specific cost discounts {"vertex_ai": 0.05} = 5% discount -cost_margin_config: Dict[ - str, Union[float, Dict[str, float]] -] = {} # Provider-specific or global cost margins. Examples: +cost_discount_config: Dict[str, float] = ( + {} +) # Provider-specific cost discounts {"vertex_ai": 0.05} = 5% discount +cost_margin_config: Dict[str, Union[float, Dict[str, float]]] = ( + {} +) # Provider-specific or global cost margins. Examples: # Percentage: {"openai": 0.10} = 10% margin # Fixed: {"openai": {"fixed_amount": 0.001}} = $0.001 per request # Global: {"global": 0.05} = 5% global margin on all providers @@ -1309,12 +1330,12 @@ from . import rag from .types.llms.custom_llm import CustomLLMItem custom_provider_map: List[CustomLLMItem] = [] -_custom_providers: List[ - str -] = [] # internal helper util, used to track names of custom providers -disable_hf_tokenizer_download: Optional[ - bool -] = None # disable huggingface tokenizer download. Defaults to openai clk100 +_custom_providers: List[str] = ( + [] +) # internal helper util, used to track names of custom providers +disable_hf_tokenizer_download: Optional[bool] = ( + None # disable huggingface tokenizer download. Defaults to openai clk100 +) global_disable_no_log_param: bool = False ### CLI UTILITIES ### diff --git a/litellm/integrations/prometheus.py b/litellm/integrations/prometheus.py index 90306c11a42..adb5e8bfc56 100644 --- a/litellm/integrations/prometheus.py +++ b/litellm/integrations/prometheus.py @@ -74,6 +74,7 @@ class PrometheusLogger(CustomLogger): # Always initialize label_filters, even for non-premium users self.label_filters = self._parse_prometheus_config() + self._parse_exclude_config() # Create metric factory functions self._counter_factory = self._create_metric_factory(Counter) @@ -103,14 +104,14 @@ class PrometheusLogger(CustomLogger): labelnames=self.get_labels_for_metric( "litellm_request_total_latency_metric" ), - buckets=LATENCY_BUCKETS, + buckets=self._get_latency_buckets(), ) self.litellm_llm_api_latency_metric = self._histogram_factory( "litellm_llm_api_latency_metric", "Total latency (seconds) for a models LLM API call", labelnames=self.get_labels_for_metric("litellm_llm_api_latency_metric"), - buckets=LATENCY_BUCKETS, + buckets=self._get_latency_buckets(), ) self.litellm_llm_api_time_to_first_token_metric = self._histogram_factory( @@ -126,7 +127,7 @@ class PrometheusLogger(CustomLogger): labelnames=self.get_labels_for_metric( "litellm_llm_api_time_to_first_token_metric" ), - buckets=LATENCY_BUCKETS, + buckets=self._get_latency_buckets(), ) # Counter for spend @@ -278,7 +279,7 @@ class PrometheusLogger(CustomLogger): labelnames=self.get_labels_for_metric( "litellm_overhead_latency_metric" ), - buckets=LATENCY_BUCKETS, + buckets=self._get_latency_buckets(), ) # Request queue time metric @@ -288,7 +289,7 @@ class PrometheusLogger(CustomLogger): labelnames=self.get_labels_for_metric( "litellm_request_queue_time_seconds" ), - buckets=LATENCY_BUCKETS, + buckets=self._get_latency_buckets(), ) # Guardrail metrics @@ -296,7 +297,7 @@ class PrometheusLogger(CustomLogger): "litellm_guardrail_latency_seconds", "Latency (seconds) for guardrail execution", labelnames=["guardrail_name", "status", "error_type", "hook_type"], - buckets=LATENCY_BUCKETS, + buckets=self._get_latency_buckets(), ) self.litellm_guardrail_errors_total = self._counter_factory( @@ -444,6 +445,26 @@ class PrometheusLogger(CustomLogger): print_verbose(f"Got exception on init prometheus client {str(e)}") raise e + def _get_latency_buckets(self) -> tuple: + """Return latency buckets to use for histogram metrics. + + Uses ``litellm.prometheus_latency_buckets`` when set, falling back to the + module-level ``LATENCY_BUCKETS`` constant. + """ + import litellm + + return litellm.prometheus_latency_buckets or LATENCY_BUCKETS + + def _parse_exclude_config(self) -> None: + """Populate self.excluded_metrics and self.excluded_labels from litellm module vars.""" + import litellm + + raw_metrics = litellm.prometheus_exclude_metrics or [] + raw_labels = litellm.prometheus_exclude_labels or [] + + self.excluded_metrics: set = set(raw_metrics) + self.excluded_labels: set = set(raw_labels) + def _parse_prometheus_config(self) -> Dict[str, List[str]]: """Parse prometheus metrics configuration for label filtering and enabled metrics""" import litellm @@ -835,7 +856,11 @@ class PrometheusLogger(CustomLogger): def _is_metric_enabled(self, metric_name: str) -> bool: """Check if a metric is enabled based on configuration""" - # If no specific configuration is provided, enable all metrics (default behavior) + # Check exclude list first — excluded metrics are always disabled + if hasattr(self, "excluded_metrics") and metric_name in self.excluded_metrics: + return False + + # If no specific include configuration is provided, enable all metrics (default behavior) if not hasattr(self, "enabled_metrics"): return True @@ -870,17 +895,17 @@ class PrometheusLogger(CustomLogger): # If no label filtering is configured for this metric, use default labels if metric_name not in self.label_filters: - return default_labels + labels = default_labels + else: + # Return intersection of configured and default labels to ensure we only use valid labels + configured_labels = self.label_filters[metric_name] + labels = [label for label in default_labels if label in configured_labels] - # Get configured labels for this metric - configured_labels = self.label_filters[metric_name] + # Strip globally excluded labels + if hasattr(self, "excluded_labels") and self.excluded_labels: + labels = [label for label in labels if label not in self.excluded_labels] - # Return intersection of configured and default labels to ensure we only use valid labels - filtered_labels = [ - label for label in default_labels if label in configured_labels - ] - - return filtered_labels + return labels async def async_log_success_event(self, kwargs, response_obj, start_time, end_time): # Define prometheus client @@ -978,9 +1003,11 @@ class PrometheusLogger(CustomLogger): ), client_ip=standard_logging_payload["metadata"].get("requester_ip_address"), user_agent=standard_logging_payload["metadata"].get("user_agent"), - stream=str(standard_logging_payload.get("stream")) - if litellm.prometheus_emit_stream_label - else None, + stream=( + str(standard_logging_payload.get("stream")) + if litellm.prometheus_emit_stream_label + else None + ), ) if ( @@ -1633,9 +1660,11 @@ class PrometheusLogger(CustomLogger): client_ip=_metadata.get("requester_ip_address"), user_agent=_metadata.get("user_agent"), model_id=model_id, - stream=str(request_data.get("stream")) - if litellm.prometheus_emit_stream_label - else None, + stream=( + str(request_data.get("stream")) + if litellm.prometheus_emit_stream_label + else None + ), ) _labels = prometheus_label_factory( supported_enum_labels=self.get_labels_for_metric( @@ -1959,9 +1988,9 @@ class PrometheusLogger(CustomLogger): ): try: verbose_logger.debug("setting remaining tokens requests metric") - standard_logging_payload: Optional[ - StandardLoggingPayload - ] = request_kwargs.get("standard_logging_object") + standard_logging_payload: Optional[StandardLoggingPayload] = ( + request_kwargs.get("standard_logging_object") + ) if standard_logging_payload is None: return @@ -2473,9 +2502,7 @@ class PrometheusLogger(CustomLogger): ) return - async def fetch_keys( - page_size: int, page: int - ) -> Tuple[ + async def fetch_keys(page_size: int, page: int) -> Tuple[ List[Union[str, UserAPIKeyAuth, LiteLLM_DeletedVerificationToken]], Optional[int], ]: @@ -2999,10 +3026,10 @@ class PrometheusLogger(CustomLogger): from litellm.constants import PROMETHEUS_BUDGET_METRICS_REFRESH_INTERVAL_MINUTES from litellm.integrations.custom_logger import CustomLogger - prometheus_loggers: List[ - CustomLogger - ] = litellm.logging_callback_manager.get_custom_loggers_for_type( - callback_type=PrometheusLogger + prometheus_loggers: List[CustomLogger] = ( + litellm.logging_callback_manager.get_custom_loggers_for_type( + callback_type=PrometheusLogger + ) ) # we need to get the initialized prometheus logger instance(s) and call logger.initialize_remaining_budget_metrics() on them verbose_logger.debug("found %s prometheus loggers", len(prometheus_loggers))