feat(prometheus): configurable latency buckets, exclude_metrics, exclude_labels

This commit is contained in:
Ishaan Jaffer 2026-03-23 11:47:44 -07:00
parent 742e176611
commit 1270ef7be0
3 changed files with 191 additions and 86 deletions

View file

@ -1,3 +1,60 @@
# Prometheus Metrics Configuration
## Custom Latency Buckets
By default, LiteLLM uses a fixed set of histogram buckets for all latency metrics. You can override them with your own tuple of bucket boundaries.
```python
import litellm
litellm.prometheus_latency_buckets = (0.05, 0.1, 0.25, 0.5, 1.0, 2.0, 5.0, float("inf"))
```
This applies to all histogram metrics:
- `litellm_request_total_latency_metric`
- `litellm_llm_api_latency_metric`
- `litellm_llm_api_time_to_first_token_metric`
- `litellm_overhead_latency_metric`
- `litellm_request_queue_time_seconds`
- `litellm_guardrail_latency_seconds`
> Set this **before** the Prometheus logger is initialized (i.e. before the proxy starts).
---
## Exclude Metrics
Disable specific metrics entirely so they are never registered or emitted.
```python
import litellm
litellm.prometheus_exclude_metrics = [
"litellm_overhead_latency_metric",
"litellm_request_queue_time_seconds",
]
```
Any metric in this list is replaced by a no-op — no Prometheus series will be created for it.
---
## Exclude Labels
Strip specific label dimensions from **all** metrics. Useful for reducing cardinality.
```python
import litellm
litellm.prometheus_exclude_labels = ["end_user", "user_agent", "client_ip"]
```
The label will be removed from every metric that would normally carry it.
> `prometheus_exclude_labels` is applied **after** any `prometheus_metrics_config` include-label filtering.
---
# 💸 GET Daily Spend, Usage Metrics # 💸 GET Daily Spend, Usage Metrics
## Request Format ## Request Format

View file

@ -167,12 +167,12 @@ prometheus_initialize_budget_metrics: Optional[bool] = False
require_auth_for_metrics_endpoint: Optional[bool] = False require_auth_for_metrics_endpoint: Optional[bool] = False
argilla_batch_size: Optional[int] = None argilla_batch_size: Optional[int] = None
datadog_use_v1: Optional[bool] = False # if you want to use v1 datadog logged payload. datadog_use_v1: Optional[bool] = False # if you want to use v1 datadog logged payload.
gcs_pub_sub_use_v1: Optional[ gcs_pub_sub_use_v1: Optional[bool] = (
bool False # if you want to use v1 gcs pubsub logged payload
] = False # if you want to use v1 gcs pubsub logged payload )
generic_api_use_v1: Optional[ generic_api_use_v1: Optional[bool] = (
bool False # if you want to use v1 generic api logged payload
] = False # if you want to use v1 generic api logged payload )
argilla_transformation_object: Optional[Dict[str, Any]] = None argilla_transformation_object: Optional[Dict[str, Any]] = None
_async_input_callback: List[ _async_input_callback: List[
Union[str, Callable, "CustomLogger"] Union[str, Callable, "CustomLogger"]
@ -192,25 +192,25 @@ _async_failure_callback: List[
pre_call_rules: List[Callable] = [] pre_call_rules: List[Callable] = []
post_call_rules: List[Callable] = [] post_call_rules: List[Callable] = []
turn_off_message_logging: Optional[bool] = False turn_off_message_logging: Optional[bool] = False
standard_logging_payload_excluded_fields: Optional[ standard_logging_payload_excluded_fields: Optional[List[str]] = (
List[str] None # Fields to exclude from StandardLoggingPayload before callbacks receive it
] = None # Fields to exclude from StandardLoggingPayload before callbacks receive it )
log_raw_request_response: bool = False log_raw_request_response: bool = False
redact_messages_in_exceptions: Optional[bool] = False redact_messages_in_exceptions: Optional[bool] = False
redact_user_api_key_info: Optional[bool] = False redact_user_api_key_info: Optional[bool] = False
filter_invalid_headers: Optional[bool] = False filter_invalid_headers: Optional[bool] = False
add_user_information_to_llm_headers: Optional[ add_user_information_to_llm_headers: Optional[bool] = (
bool None # adds user_id, team_id, token hash (params from StandardLoggingMetadata) to request headers
] = None # adds user_id, team_id, token hash (params from StandardLoggingMetadata) to request headers )
store_audit_logs = False # Enterprise feature, allow users to see audit logs store_audit_logs = False # Enterprise feature, allow users to see audit logs
### end of callbacks ############# ### end of callbacks #############
email: Optional[ email: Optional[str] = (
str None # Not used anymore, will be removed in next MAJOR release - https://github.com/BerriAI/litellm/discussions/648
] = None # Not used anymore, will be removed in next MAJOR release - https://github.com/BerriAI/litellm/discussions/648 )
token: Optional[ token: Optional[str] = (
str None # Not used anymore, will be removed in next MAJOR release - https://github.com/BerriAI/litellm/discussions/648
] = None # Not used anymore, will be removed in next MAJOR release - https://github.com/BerriAI/litellm/discussions/648 )
telemetry = True telemetry = True
max_tokens: int = DEFAULT_MAX_TOKENS # OpenAI Defaults max_tokens: int = DEFAULT_MAX_TOKENS # OpenAI Defaults
drop_params = bool(os.getenv("LITELLM_DROP_PARAMS", False)) drop_params = bool(os.getenv("LITELLM_DROP_PARAMS", False))
@ -272,9 +272,9 @@ use_client: bool = False
ssl_verify: Union[str, bool] = True ssl_verify: Union[str, bool] = True
ssl_security_level: Optional[str] = None ssl_security_level: Optional[str] = None
ssl_certificate: Optional[str] = None ssl_certificate: Optional[str] = None
ssl_ecdh_curve: Optional[ ssl_ecdh_curve: Optional[str] = (
str None # Set to 'X25519' to disable PQC and improve performance
] = None # Set to 'X25519' to disable PQC and improve performance )
disable_streaming_logging: bool = False disable_streaming_logging: bool = False
disable_token_counter: bool = False disable_token_counter: bool = False
disable_add_transform_inline_image_block: bool = False disable_add_transform_inline_image_block: bool = False
@ -327,20 +327,24 @@ enable_loadbalancing_on_batch_endpoints: Optional[bool] = None
enable_caching_on_provider_specific_optional_params: bool = ( enable_caching_on_provider_specific_optional_params: bool = (
False # feature-flag for caching on optional params - e.g. 'top_k' False # feature-flag for caching on optional params - e.g. 'top_k'
) )
caching: bool = False # Not used anymore, will be removed in next MAJOR release - https://github.com/BerriAI/litellm/discussions/648 caching: bool = (
caching_with_models: bool = False # # Not used anymore, will be removed in next MAJOR release - https://github.com/BerriAI/litellm/discussions/648 False # Not used anymore, will be removed in next MAJOR release - https://github.com/BerriAI/litellm/discussions/648
cache: Optional[ )
"Cache" caching_with_models: bool = (
] = None # cache object <- use this - https://docs.litellm.ai/docs/caching False # # Not used anymore, will be removed in next MAJOR release - https://github.com/BerriAI/litellm/discussions/648
)
cache: Optional["Cache"] = (
None # cache object <- use this - https://docs.litellm.ai/docs/caching
)
default_in_memory_ttl: Optional[float] = None default_in_memory_ttl: Optional[float] = None
default_redis_ttl: Optional[float] = None default_redis_ttl: Optional[float] = None
default_redis_batch_cache_expiry: Optional[float] = None default_redis_batch_cache_expiry: Optional[float] = None
model_alias_map: Dict[str, str] = {} model_alias_map: Dict[str, str] = {}
model_group_settings: Optional["ModelGroupSettings"] = None model_group_settings: Optional["ModelGroupSettings"] = None
max_budget: float = 0.0 # set the max budget across all providers max_budget: float = 0.0 # set the max budget across all providers
budget_duration: Optional[ budget_duration: Optional[str] = (
str None # proxy only - resets budget after fixed duration. You can set duration as seconds ("30s"), minutes ("30m"), hours ("30h"), days ("30d").
] = None # proxy only - resets budget after fixed duration. You can set duration as seconds ("30s"), minutes ("30m"), hours ("30h"), days ("30d"). )
default_soft_budget: float = ( default_soft_budget: float = (
DEFAULT_SOFT_BUDGET # by default all litellm proxy keys have a soft budget of 50.0 DEFAULT_SOFT_BUDGET # by default all litellm proxy keys have a soft budget of 50.0
) )
@ -349,7 +353,9 @@ forward_traceparent_to_llm_provider: bool = False
_current_cost = 0.0 # private variable, used if max budget is set _current_cost = 0.0 # private variable, used if max budget is set
error_logs: Dict = {} error_logs: Dict = {}
add_function_to_prompt: bool = False # if function calling not supported by api, append function call details to system prompt add_function_to_prompt: bool = (
False # if function calling not supported by api, append function call details to system prompt
)
client_session: Optional[httpx.Client] = None client_session: Optional[httpx.Client] = None
aclient_session: Optional[httpx.AsyncClient] = None aclient_session: Optional[httpx.AsyncClient] = None
model_fallbacks: Optional[List] = None # Deprecated for 'litellm.fallbacks' model_fallbacks: Optional[List] = None # Deprecated for 'litellm.fallbacks'
@ -392,11 +398,22 @@ enable_end_user_cost_tracking_prometheus_only: Optional[bool] = None
custom_prometheus_metadata_labels: List[str] = [] custom_prometheus_metadata_labels: List[str] = []
custom_prometheus_tags: List[str] = [] custom_prometheus_tags: List[str] = []
prometheus_metrics_config: Optional[List] = None prometheus_metrics_config: Optional[List] = None
prometheus_latency_buckets: Optional[Tuple] = (
None # override default LATENCY_BUCKETS for histogram metrics
)
prometheus_exclude_metrics: Optional[List[str]] = (
None # metric names to disable entirely
)
prometheus_exclude_labels: Optional[List[str]] = (
None # label names to strip from all metrics
)
prometheus_emit_stream_label: bool = False prometheus_emit_stream_label: bool = False
disable_add_prefix_to_prompt: bool = ( disable_add_prefix_to_prompt: bool = (
False # used by anthropic, to disable adding prefix to prompt False # used by anthropic, to disable adding prefix to prompt
) )
disable_copilot_system_to_assistant: bool = False # If false (default), converts all 'system' role messages to 'assistant' for GitHub Copilot compatibility. Set to true to disable this behavior. disable_copilot_system_to_assistant: bool = (
False # If false (default), converts all 'system' role messages to 'assistant' for GitHub Copilot compatibility. Set to true to disable this behavior.
)
public_mcp_servers: Optional[List[str]] = None public_mcp_servers: Optional[List[str]] = None
public_model_groups: Optional[List[str]] = None public_model_groups: Optional[List[str]] = None
public_agent_groups: Optional[List[str]] = None public_agent_groups: Optional[List[str]] = None
@ -405,9 +422,9 @@ public_agent_groups: Optional[List[str]] = None
# Old format: { "displayName": "url" } (for backward compatibility) # Old format: { "displayName": "url" } (for backward compatibility)
public_model_groups_links: Dict[str, Union[str, Dict[str, Any]]] = {} public_model_groups_links: Dict[str, Union[str, Dict[str, Any]]] = {}
#### REQUEST PRIORITIZATION ####### #### REQUEST PRIORITIZATION #######
priority_reservation: Optional[ priority_reservation: Optional[Dict[str, Union[float, "PriorityReservationDict"]]] = (
Dict[str, Union[float, "PriorityReservationDict"]] None
] = None )
# priority_reservation_settings is lazy-loaded via __getattr__ # priority_reservation_settings is lazy-loaded via __getattr__
# Only declare for type checking - at runtime __getattr__ handles it # Only declare for type checking - at runtime __getattr__ handles it
if TYPE_CHECKING: if TYPE_CHECKING:
@ -415,13 +432,17 @@ if TYPE_CHECKING:
######## Networking Settings ######## ######## Networking Settings ########
use_aiohttp_transport: bool = True # Older variable, aiohttp is now the default. use disable_aiohttp_transport instead. use_aiohttp_transport: bool = (
True # Older variable, aiohttp is now the default. use disable_aiohttp_transport instead.
)
aiohttp_trust_env: bool = False # set to true to use HTTP_ Proxy settings aiohttp_trust_env: bool = False # set to true to use HTTP_ Proxy settings
disable_aiohttp_transport: bool = False # Set this to true to use httpx instead disable_aiohttp_transport: bool = False # Set this to true to use httpx instead
disable_aiohttp_trust_env: bool = ( disable_aiohttp_trust_env: bool = (
False # When False, aiohttp will respect HTTP(S)_PROXY env vars False # When False, aiohttp will respect HTTP(S)_PROXY env vars
) )
force_ipv4: bool = False # when True, litellm will force ipv4 for all LLM requests. Some users have seen httpx ConnectionError when using ipv6. force_ipv4: bool = (
False # when True, litellm will force ipv4 for all LLM requests. Some users have seen httpx ConnectionError when using ipv6.
)
network_mock: bool = False # When True, use mock transport — no real network calls network_mock: bool = False # When True, use mock transport — no real network calls
####### STOP SEQUENCE LIMIT ####### ####### STOP SEQUENCE LIMIT #######
@ -436,13 +457,13 @@ context_window_fallbacks: Optional[List] = None
content_policy_fallbacks: Optional[List] = None content_policy_fallbacks: Optional[List] = None
allowed_fails: int = 3 allowed_fails: int = 3
allow_dynamic_callback_disabling: bool = True allow_dynamic_callback_disabling: bool = True
num_retries_per_request: Optional[ num_retries_per_request: Optional[int] = (
int None # for the request overall (incl. fallbacks + model retries)
] = None # for the request overall (incl. fallbacks + model retries) )
####### SECRET MANAGERS ##################### ####### SECRET MANAGERS #####################
secret_manager_client: Optional[ secret_manager_client: Optional[Any] = (
Any None # list of instantiated key management clients - e.g. azure kv, infisical, etc.
] = None # list of instantiated key management clients - e.g. azure kv, infisical, etc. )
_google_kms_resource_name: Optional[str] = None _google_kms_resource_name: Optional[str] = None
_key_management_system: Optional["KeyManagementSystem"] = None _key_management_system: Optional["KeyManagementSystem"] = None
# Note: KeyManagementSettings must be eagerly imported because _key_management_settings # Note: KeyManagementSettings must be eagerly imported because _key_management_settings
@ -455,12 +476,12 @@ output_parse_pii: bool = False
from litellm.litellm_core_utils.get_model_cost_map import get_model_cost_map from litellm.litellm_core_utils.get_model_cost_map import get_model_cost_map
model_cost = get_model_cost_map(url=model_cost_map_url) model_cost = get_model_cost_map(url=model_cost_map_url)
cost_discount_config: Dict[ cost_discount_config: Dict[str, float] = (
str, float {}
] = {} # Provider-specific cost discounts {"vertex_ai": 0.05} = 5% discount ) # Provider-specific cost discounts {"vertex_ai": 0.05} = 5% discount
cost_margin_config: Dict[ cost_margin_config: Dict[str, Union[float, Dict[str, float]]] = (
str, Union[float, Dict[str, float]] {}
] = {} # Provider-specific or global cost margins. Examples: ) # Provider-specific or global cost margins. Examples:
# Percentage: {"openai": 0.10} = 10% margin # Percentage: {"openai": 0.10} = 10% margin
# Fixed: {"openai": {"fixed_amount": 0.001}} = $0.001 per request # Fixed: {"openai": {"fixed_amount": 0.001}} = $0.001 per request
# Global: {"global": 0.05} = 5% global margin on all providers # Global: {"global": 0.05} = 5% global margin on all providers
@ -1309,12 +1330,12 @@ from . import rag
from .types.llms.custom_llm import CustomLLMItem from .types.llms.custom_llm import CustomLLMItem
custom_provider_map: List[CustomLLMItem] = [] custom_provider_map: List[CustomLLMItem] = []
_custom_providers: List[ _custom_providers: List[str] = (
str []
] = [] # internal helper util, used to track names of custom providers ) # internal helper util, used to track names of custom providers
disable_hf_tokenizer_download: Optional[ disable_hf_tokenizer_download: Optional[bool] = (
bool None # disable huggingface tokenizer download. Defaults to openai clk100
] = None # disable huggingface tokenizer download. Defaults to openai clk100 )
global_disable_no_log_param: bool = False global_disable_no_log_param: bool = False
### CLI UTILITIES ### ### CLI UTILITIES ###

View file

@ -74,6 +74,7 @@ class PrometheusLogger(CustomLogger):
# Always initialize label_filters, even for non-premium users # Always initialize label_filters, even for non-premium users
self.label_filters = self._parse_prometheus_config() self.label_filters = self._parse_prometheus_config()
self._parse_exclude_config()
# Create metric factory functions # Create metric factory functions
self._counter_factory = self._create_metric_factory(Counter) self._counter_factory = self._create_metric_factory(Counter)
@ -103,14 +104,14 @@ class PrometheusLogger(CustomLogger):
labelnames=self.get_labels_for_metric( labelnames=self.get_labels_for_metric(
"litellm_request_total_latency_metric" "litellm_request_total_latency_metric"
), ),
buckets=LATENCY_BUCKETS, buckets=self._get_latency_buckets(),
) )
self.litellm_llm_api_latency_metric = self._histogram_factory( self.litellm_llm_api_latency_metric = self._histogram_factory(
"litellm_llm_api_latency_metric", "litellm_llm_api_latency_metric",
"Total latency (seconds) for a models LLM API call", "Total latency (seconds) for a models LLM API call",
labelnames=self.get_labels_for_metric("litellm_llm_api_latency_metric"), labelnames=self.get_labels_for_metric("litellm_llm_api_latency_metric"),
buckets=LATENCY_BUCKETS, buckets=self._get_latency_buckets(),
) )
self.litellm_llm_api_time_to_first_token_metric = self._histogram_factory( self.litellm_llm_api_time_to_first_token_metric = self._histogram_factory(
@ -126,7 +127,7 @@ class PrometheusLogger(CustomLogger):
labelnames=self.get_labels_for_metric( labelnames=self.get_labels_for_metric(
"litellm_llm_api_time_to_first_token_metric" "litellm_llm_api_time_to_first_token_metric"
), ),
buckets=LATENCY_BUCKETS, buckets=self._get_latency_buckets(),
) )
# Counter for spend # Counter for spend
@ -278,7 +279,7 @@ class PrometheusLogger(CustomLogger):
labelnames=self.get_labels_for_metric( labelnames=self.get_labels_for_metric(
"litellm_overhead_latency_metric" "litellm_overhead_latency_metric"
), ),
buckets=LATENCY_BUCKETS, buckets=self._get_latency_buckets(),
) )
# Request queue time metric # Request queue time metric
@ -288,7 +289,7 @@ class PrometheusLogger(CustomLogger):
labelnames=self.get_labels_for_metric( labelnames=self.get_labels_for_metric(
"litellm_request_queue_time_seconds" "litellm_request_queue_time_seconds"
), ),
buckets=LATENCY_BUCKETS, buckets=self._get_latency_buckets(),
) )
# Guardrail metrics # Guardrail metrics
@ -296,7 +297,7 @@ class PrometheusLogger(CustomLogger):
"litellm_guardrail_latency_seconds", "litellm_guardrail_latency_seconds",
"Latency (seconds) for guardrail execution", "Latency (seconds) for guardrail execution",
labelnames=["guardrail_name", "status", "error_type", "hook_type"], labelnames=["guardrail_name", "status", "error_type", "hook_type"],
buckets=LATENCY_BUCKETS, buckets=self._get_latency_buckets(),
) )
self.litellm_guardrail_errors_total = self._counter_factory( self.litellm_guardrail_errors_total = self._counter_factory(
@ -444,6 +445,26 @@ class PrometheusLogger(CustomLogger):
print_verbose(f"Got exception on init prometheus client {str(e)}") print_verbose(f"Got exception on init prometheus client {str(e)}")
raise e raise e
def _get_latency_buckets(self) -> tuple:
"""Return latency buckets to use for histogram metrics.
Uses ``litellm.prometheus_latency_buckets`` when set, falling back to the
module-level ``LATENCY_BUCKETS`` constant.
"""
import litellm
return litellm.prometheus_latency_buckets or LATENCY_BUCKETS
def _parse_exclude_config(self) -> None:
"""Populate self.excluded_metrics and self.excluded_labels from litellm module vars."""
import litellm
raw_metrics = litellm.prometheus_exclude_metrics or []
raw_labels = litellm.prometheus_exclude_labels or []
self.excluded_metrics: set = set(raw_metrics)
self.excluded_labels: set = set(raw_labels)
def _parse_prometheus_config(self) -> Dict[str, List[str]]: def _parse_prometheus_config(self) -> Dict[str, List[str]]:
"""Parse prometheus metrics configuration for label filtering and enabled metrics""" """Parse prometheus metrics configuration for label filtering and enabled metrics"""
import litellm import litellm
@ -835,7 +856,11 @@ class PrometheusLogger(CustomLogger):
def _is_metric_enabled(self, metric_name: str) -> bool: def _is_metric_enabled(self, metric_name: str) -> bool:
"""Check if a metric is enabled based on configuration""" """Check if a metric is enabled based on configuration"""
# If no specific configuration is provided, enable all metrics (default behavior) # Check exclude list first — excluded metrics are always disabled
if hasattr(self, "excluded_metrics") and metric_name in self.excluded_metrics:
return False
# If no specific include configuration is provided, enable all metrics (default behavior)
if not hasattr(self, "enabled_metrics"): if not hasattr(self, "enabled_metrics"):
return True return True
@ -870,17 +895,17 @@ class PrometheusLogger(CustomLogger):
# If no label filtering is configured for this metric, use default labels # If no label filtering is configured for this metric, use default labels
if metric_name not in self.label_filters: if metric_name not in self.label_filters:
return default_labels labels = default_labels
else:
# Return intersection of configured and default labels to ensure we only use valid labels
configured_labels = self.label_filters[metric_name]
labels = [label for label in default_labels if label in configured_labels]
# Get configured labels for this metric # Strip globally excluded labels
configured_labels = self.label_filters[metric_name] if hasattr(self, "excluded_labels") and self.excluded_labels:
labels = [label for label in labels if label not in self.excluded_labels]
# Return intersection of configured and default labels to ensure we only use valid labels return labels
filtered_labels = [
label for label in default_labels if label in configured_labels
]
return filtered_labels
async def async_log_success_event(self, kwargs, response_obj, start_time, end_time): async def async_log_success_event(self, kwargs, response_obj, start_time, end_time):
# Define prometheus client # Define prometheus client
@ -978,9 +1003,11 @@ class PrometheusLogger(CustomLogger):
), ),
client_ip=standard_logging_payload["metadata"].get("requester_ip_address"), client_ip=standard_logging_payload["metadata"].get("requester_ip_address"),
user_agent=standard_logging_payload["metadata"].get("user_agent"), user_agent=standard_logging_payload["metadata"].get("user_agent"),
stream=str(standard_logging_payload.get("stream")) stream=(
if litellm.prometheus_emit_stream_label str(standard_logging_payload.get("stream"))
else None, if litellm.prometheus_emit_stream_label
else None
),
) )
if ( if (
@ -1633,9 +1660,11 @@ class PrometheusLogger(CustomLogger):
client_ip=_metadata.get("requester_ip_address"), client_ip=_metadata.get("requester_ip_address"),
user_agent=_metadata.get("user_agent"), user_agent=_metadata.get("user_agent"),
model_id=model_id, model_id=model_id,
stream=str(request_data.get("stream")) stream=(
if litellm.prometheus_emit_stream_label str(request_data.get("stream"))
else None, if litellm.prometheus_emit_stream_label
else None
),
) )
_labels = prometheus_label_factory( _labels = prometheus_label_factory(
supported_enum_labels=self.get_labels_for_metric( supported_enum_labels=self.get_labels_for_metric(
@ -1959,9 +1988,9 @@ class PrometheusLogger(CustomLogger):
): ):
try: try:
verbose_logger.debug("setting remaining tokens requests metric") verbose_logger.debug("setting remaining tokens requests metric")
standard_logging_payload: Optional[ standard_logging_payload: Optional[StandardLoggingPayload] = (
StandardLoggingPayload request_kwargs.get("standard_logging_object")
] = request_kwargs.get("standard_logging_object") )
if standard_logging_payload is None: if standard_logging_payload is None:
return return
@ -2473,9 +2502,7 @@ class PrometheusLogger(CustomLogger):
) )
return return
async def fetch_keys( async def fetch_keys(page_size: int, page: int) -> Tuple[
page_size: int, page: int
) -> Tuple[
List[Union[str, UserAPIKeyAuth, LiteLLM_DeletedVerificationToken]], List[Union[str, UserAPIKeyAuth, LiteLLM_DeletedVerificationToken]],
Optional[int], Optional[int],
]: ]:
@ -2999,10 +3026,10 @@ class PrometheusLogger(CustomLogger):
from litellm.constants import PROMETHEUS_BUDGET_METRICS_REFRESH_INTERVAL_MINUTES from litellm.constants import PROMETHEUS_BUDGET_METRICS_REFRESH_INTERVAL_MINUTES
from litellm.integrations.custom_logger import CustomLogger from litellm.integrations.custom_logger import CustomLogger
prometheus_loggers: List[ prometheus_loggers: List[CustomLogger] = (
CustomLogger litellm.logging_callback_manager.get_custom_loggers_for_type(
] = litellm.logging_callback_manager.get_custom_loggers_for_type( callback_type=PrometheusLogger
callback_type=PrometheusLogger )
) )
# we need to get the initialized prometheus logger instance(s) and call logger.initialize_remaining_budget_metrics() on them # we need to get the initialized prometheus logger instance(s) and call logger.initialize_remaining_budget_metrics() on them
verbose_logger.debug("found %s prometheus loggers", len(prometheus_loggers)) verbose_logger.debug("found %s prometheus loggers", len(prometheus_loggers))