mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
This commit deletes the AuthMetrics class and its associated methods, which were responsible for tracking combined_view SQL query metrics. The PrometheusLogger integration has been updated to remove references to these metrics, streamlining the codebase. Additionally, minor whitespace adjustments were made in the cache coordinator for consistency.
3689 lines
140 KiB
Python
3689 lines
140 KiB
Python
# used for /metrics endpoint on LiteLLM Proxy
|
|
#### What this does ####
|
|
# On success, log events to Prometheus
|
|
from __future__ import annotations
|
|
|
|
import asyncio
|
|
import os
|
|
import sys
|
|
from datetime import datetime, timedelta
|
|
from typing import (
|
|
TYPE_CHECKING,
|
|
Any,
|
|
Awaitable,
|
|
Callable,
|
|
Dict,
|
|
List,
|
|
Literal,
|
|
Optional,
|
|
Sequence,
|
|
Tuple,
|
|
Union,
|
|
cast,
|
|
)
|
|
|
|
import litellm
|
|
from litellm._logging import print_verbose, verbose_logger
|
|
from litellm.integrations.custom_logger import CustomLogger
|
|
from litellm.integrations.prometheus_helpers import (
|
|
PrometheusLabelFactoryContext,
|
|
_get_cached_end_user_id_for_cost_tracking,
|
|
)
|
|
from litellm.litellm_core_utils.core_helpers import (
|
|
get_litellm_metadata_from_kwargs,
|
|
get_metadata_variable_name_from_kwargs,
|
|
)
|
|
from litellm.proxy._types import (
|
|
LiteLLM_DeletedVerificationToken,
|
|
LiteLLM_TeamTable,
|
|
LiteLLM_UserTable,
|
|
UserAPIKeyAuth,
|
|
)
|
|
from litellm.types.integrations.prometheus import *
|
|
from litellm.types.integrations.prometheus import (
|
|
_sanitize_prometheus_label_name,
|
|
_sanitize_prometheus_label_value,
|
|
)
|
|
from litellm.types.utils import StandardLoggingPayload
|
|
|
|
if TYPE_CHECKING:
|
|
from apscheduler.schedulers.asyncio import AsyncIOScheduler
|
|
else:
|
|
AsyncIOScheduler = Any
|
|
|
|
|
|
class PrometheusLogger(CustomLogger):
|
|
# Class variables or attributes
|
|
|
|
@staticmethod
|
|
def get_instance() -> Optional["PrometheusLogger"]:
|
|
"""Find the PrometheusLogger instance from litellm.callbacks, if registered."""
|
|
import litellm
|
|
|
|
for cb in litellm.callbacks:
|
|
if isinstance(cb, PrometheusLogger):
|
|
return cb
|
|
return None
|
|
|
|
def __init__( # noqa: PLR0915
|
|
self,
|
|
**kwargs,
|
|
):
|
|
try:
|
|
from prometheus_client import Counter, Gauge, Histogram
|
|
|
|
# Always initialize label_filters, even for non-premium users
|
|
self.label_filters = self._parse_prometheus_config()
|
|
|
|
_custom_buckets = litellm.prometheus_latency_buckets
|
|
self.latency_buckets = (
|
|
tuple(_custom_buckets)
|
|
if _custom_buckets is not None
|
|
else LATENCY_BUCKETS
|
|
)
|
|
|
|
# Create metric factory functions
|
|
self._counter_factory = self._create_metric_factory(Counter)
|
|
self._gauge_factory = self._create_metric_factory(Gauge)
|
|
self._histogram_factory = self._create_metric_factory(Histogram)
|
|
|
|
self.litellm_proxy_failed_requests_metric = self._counter_factory(
|
|
name="litellm_proxy_failed_requests_metric",
|
|
documentation="Total number of failed responses from proxy - the client did not get a success response from litellm proxy",
|
|
labelnames=self.get_labels_for_metric(
|
|
"litellm_proxy_failed_requests_metric"
|
|
),
|
|
)
|
|
|
|
self.litellm_proxy_total_requests_metric = self._counter_factory(
|
|
name="litellm_proxy_total_requests_metric",
|
|
documentation="Total number of requests made to the proxy server - track number of client side requests",
|
|
labelnames=self.get_labels_for_metric(
|
|
"litellm_proxy_total_requests_metric"
|
|
),
|
|
)
|
|
|
|
# request latency metrics
|
|
self.litellm_request_total_latency_metric = self._histogram_factory(
|
|
"litellm_request_total_latency_metric",
|
|
"Total latency (seconds) for a request to LiteLLM",
|
|
labelnames=self.get_labels_for_metric(
|
|
"litellm_request_total_latency_metric"
|
|
),
|
|
buckets=self.latency_buckets,
|
|
)
|
|
|
|
self.litellm_llm_api_latency_metric = self._histogram_factory(
|
|
"litellm_llm_api_latency_metric",
|
|
"Total latency (seconds) for a models LLM API call",
|
|
labelnames=self.get_labels_for_metric("litellm_llm_api_latency_metric"),
|
|
buckets=self.latency_buckets,
|
|
)
|
|
|
|
self.litellm_llm_api_time_to_first_token_metric = self._histogram_factory(
|
|
"litellm_llm_api_time_to_first_token_metric",
|
|
"Time to first token for a models LLM API call",
|
|
# labelnames=[
|
|
# "model",
|
|
# "hashed_api_key",
|
|
# "api_key_alias",
|
|
# "team",
|
|
# "team_alias",
|
|
# ],
|
|
labelnames=self.get_labels_for_metric(
|
|
"litellm_llm_api_time_to_first_token_metric"
|
|
),
|
|
buckets=self.latency_buckets,
|
|
)
|
|
|
|
# Counter for spend
|
|
self.litellm_spend_metric = self._counter_factory(
|
|
"litellm_spend_metric",
|
|
"Total spend on LLM requests",
|
|
labelnames=self.get_labels_for_metric("litellm_spend_metric"),
|
|
)
|
|
|
|
# Counter for total_output_tokens
|
|
self.litellm_tokens_metric = self._counter_factory(
|
|
"litellm_total_tokens_metric",
|
|
"Total number of input + output tokens from LLM requests",
|
|
labelnames=self.get_labels_for_metric("litellm_total_tokens_metric"),
|
|
)
|
|
|
|
self.litellm_input_tokens_metric = self._counter_factory(
|
|
"litellm_input_tokens_metric",
|
|
"Total number of input tokens from LLM requests",
|
|
labelnames=self.get_labels_for_metric("litellm_input_tokens_metric"),
|
|
)
|
|
|
|
self.litellm_output_tokens_metric = self._counter_factory(
|
|
"litellm_output_tokens_metric",
|
|
"Total number of output tokens from LLM requests",
|
|
labelnames=self.get_labels_for_metric("litellm_output_tokens_metric"),
|
|
)
|
|
|
|
# Remaining Budget for Team
|
|
self.litellm_remaining_team_budget_metric = self._gauge_factory(
|
|
"litellm_remaining_team_budget_metric",
|
|
"Remaining budget for team",
|
|
labelnames=self.get_labels_for_metric(
|
|
"litellm_remaining_team_budget_metric"
|
|
),
|
|
)
|
|
|
|
# Max Budget for Team
|
|
self.litellm_team_max_budget_metric = self._gauge_factory(
|
|
"litellm_team_max_budget_metric",
|
|
"Maximum budget set for team",
|
|
labelnames=self.get_labels_for_metric("litellm_team_max_budget_metric"),
|
|
)
|
|
|
|
# Team Budget Reset At
|
|
self.litellm_team_budget_remaining_hours_metric = self._gauge_factory(
|
|
"litellm_team_budget_remaining_hours_metric",
|
|
"Remaining days for team budget to be reset",
|
|
labelnames=self.get_labels_for_metric(
|
|
"litellm_team_budget_remaining_hours_metric"
|
|
),
|
|
)
|
|
|
|
# Remaining Budget for Org
|
|
self.litellm_remaining_org_budget_metric = self._gauge_factory(
|
|
"litellm_remaining_org_budget_metric",
|
|
"Remaining budget for org",
|
|
labelnames=self.get_labels_for_metric(
|
|
"litellm_remaining_org_budget_metric"
|
|
),
|
|
)
|
|
|
|
# Max Budget for Org
|
|
self.litellm_org_max_budget_metric = self._gauge_factory(
|
|
"litellm_org_max_budget_metric",
|
|
"Maximum budget set for org",
|
|
labelnames=self.get_labels_for_metric("litellm_org_max_budget_metric"),
|
|
)
|
|
|
|
# Org Budget Reset At
|
|
self.litellm_org_budget_remaining_hours_metric = self._gauge_factory(
|
|
"litellm_org_budget_remaining_hours_metric",
|
|
"Remaining hours for org budget to be reset",
|
|
labelnames=self.get_labels_for_metric(
|
|
"litellm_org_budget_remaining_hours_metric"
|
|
),
|
|
)
|
|
|
|
# Remaining Budget for API Key
|
|
self.litellm_remaining_api_key_budget_metric = self._gauge_factory(
|
|
"litellm_remaining_api_key_budget_metric",
|
|
"Remaining budget for api key",
|
|
labelnames=self.get_labels_for_metric(
|
|
"litellm_remaining_api_key_budget_metric"
|
|
),
|
|
)
|
|
|
|
# Max Budget for API Key
|
|
self.litellm_api_key_max_budget_metric = self._gauge_factory(
|
|
"litellm_api_key_max_budget_metric",
|
|
"Maximum budget set for api key",
|
|
labelnames=self.get_labels_for_metric(
|
|
"litellm_api_key_max_budget_metric"
|
|
),
|
|
)
|
|
|
|
self.litellm_api_key_budget_remaining_hours_metric = self._gauge_factory(
|
|
"litellm_api_key_budget_remaining_hours_metric",
|
|
"Remaining hours for api key budget to be reset",
|
|
labelnames=self.get_labels_for_metric(
|
|
"litellm_api_key_budget_remaining_hours_metric"
|
|
),
|
|
)
|
|
|
|
# Remaining Budget for User
|
|
self.litellm_remaining_user_budget_metric = self._gauge_factory(
|
|
"litellm_remaining_user_budget_metric",
|
|
"Remaining budget for user",
|
|
labelnames=self.get_labels_for_metric(
|
|
"litellm_remaining_user_budget_metric"
|
|
),
|
|
)
|
|
|
|
# Max Budget for User
|
|
self.litellm_user_max_budget_metric = self._gauge_factory(
|
|
"litellm_user_max_budget_metric",
|
|
"Maximum budget set for user",
|
|
labelnames=self.get_labels_for_metric("litellm_user_max_budget_metric"),
|
|
)
|
|
|
|
self.litellm_user_budget_remaining_hours_metric = self._gauge_factory(
|
|
"litellm_user_budget_remaining_hours_metric",
|
|
"Remaining hours for user budget to be reset",
|
|
labelnames=self.get_labels_for_metric(
|
|
"litellm_user_budget_remaining_hours_metric"
|
|
),
|
|
)
|
|
|
|
########################################
|
|
# LiteLLM Virtual API KEY metrics
|
|
########################################
|
|
|
|
# Remaining MODEL RPM limit for API Key
|
|
self.litellm_remaining_api_key_requests_for_model = self._gauge_factory(
|
|
"litellm_remaining_api_key_requests_for_model",
|
|
"Remaining Requests API Key can make for model (model based rpm limit on key)",
|
|
labelnames=self.get_labels_for_metric(
|
|
"litellm_remaining_api_key_requests_for_model"
|
|
),
|
|
)
|
|
|
|
# Remaining MODEL TPM limit for API Key
|
|
self.litellm_remaining_api_key_tokens_for_model = self._gauge_factory(
|
|
"litellm_remaining_api_key_tokens_for_model",
|
|
"Remaining Tokens API Key can make for model (model based tpm limit on key)",
|
|
labelnames=self.get_labels_for_metric(
|
|
"litellm_remaining_api_key_tokens_for_model"
|
|
),
|
|
)
|
|
|
|
########################################
|
|
# LLM API Deployment Metrics / analytics
|
|
########################################
|
|
|
|
# Remaining Rate Limit for model
|
|
self.litellm_remaining_requests_metric = self._gauge_factory(
|
|
"litellm_remaining_requests_metric",
|
|
"LLM Deployment Analytics - remaining requests for model, returned from LLM API Provider",
|
|
labelnames=self.get_labels_for_metric(
|
|
"litellm_remaining_requests_metric"
|
|
),
|
|
)
|
|
|
|
self.litellm_remaining_tokens_metric = self._gauge_factory(
|
|
"litellm_remaining_tokens_metric",
|
|
"remaining tokens for model, returned from LLM API Provider",
|
|
labelnames=self.get_labels_for_metric(
|
|
"litellm_remaining_tokens_metric"
|
|
),
|
|
)
|
|
|
|
self.litellm_overhead_latency_metric = self._histogram_factory(
|
|
"litellm_overhead_latency_metric",
|
|
"Latency overhead (milliseconds) added by LiteLLM processing",
|
|
labelnames=self.get_labels_for_metric(
|
|
"litellm_overhead_latency_metric"
|
|
),
|
|
buckets=self.latency_buckets,
|
|
)
|
|
|
|
# Request queue time metric
|
|
self.litellm_request_queue_time_metric = self._histogram_factory(
|
|
"litellm_request_queue_time_seconds",
|
|
"Time spent in request queue before processing starts (seconds)",
|
|
labelnames=self.get_labels_for_metric(
|
|
"litellm_request_queue_time_seconds"
|
|
),
|
|
buckets=self.latency_buckets,
|
|
)
|
|
|
|
# Guardrail metrics
|
|
self.litellm_guardrail_latency_metric = self._histogram_factory(
|
|
"litellm_guardrail_latency_seconds",
|
|
"Latency (seconds) for guardrail execution",
|
|
labelnames=["guardrail_name", "status", "error_type", "hook_type"],
|
|
buckets=self.latency_buckets,
|
|
)
|
|
|
|
self.litellm_guardrail_errors_total = self._counter_factory(
|
|
"litellm_guardrail_errors_total",
|
|
"Total number of errors encountered during guardrail execution",
|
|
labelnames=["guardrail_name", "error_type", "hook_type"],
|
|
)
|
|
|
|
self.litellm_guardrail_requests_total = self._counter_factory(
|
|
"litellm_guardrail_requests_total",
|
|
"Total number of guardrail invocations",
|
|
labelnames=["guardrail_name", "status", "hook_type"],
|
|
)
|
|
# llm api provider budget metrics
|
|
self.litellm_provider_remaining_budget_metric = self._gauge_factory(
|
|
"litellm_provider_remaining_budget_metric",
|
|
"Remaining budget for provider - used when you set provider budget limits",
|
|
labelnames=["api_provider"],
|
|
)
|
|
|
|
# Metric for deployment state
|
|
self.litellm_deployment_state = self._gauge_factory(
|
|
"litellm_deployment_state",
|
|
"LLM Deployment Analytics - The state of the deployment: 0 = healthy, 1 = partial outage, 2 = complete outage",
|
|
labelnames=self.get_labels_for_metric("litellm_deployment_state"),
|
|
)
|
|
|
|
self.litellm_deployment_tpm_limit = self._gauge_factory(
|
|
"litellm_deployment_tpm_limit",
|
|
"Deployment TPM limit found in config",
|
|
labelnames=self.get_labels_for_metric("litellm_deployment_tpm_limit"),
|
|
)
|
|
|
|
self.litellm_deployment_rpm_limit = self._gauge_factory(
|
|
"litellm_deployment_rpm_limit",
|
|
"Deployment RPM limit found in config",
|
|
labelnames=self.get_labels_for_metric("litellm_deployment_rpm_limit"),
|
|
)
|
|
|
|
self.litellm_deployment_cooled_down = self._counter_factory(
|
|
"litellm_deployment_cooled_down",
|
|
"LLM Deployment Analytics - Number of times a deployment has been cooled down by LiteLLM load balancing logic. exception_status is the status of the exception that caused the deployment to be cooled down",
|
|
# labelnames=_logged_llm_labels + [EXCEPTION_STATUS],
|
|
labelnames=self.get_labels_for_metric("litellm_deployment_cooled_down"),
|
|
)
|
|
|
|
self.litellm_deployment_success_responses = self._counter_factory(
|
|
name="litellm_deployment_success_responses",
|
|
documentation="LLM Deployment Analytics - Total number of successful LLM API calls via litellm",
|
|
labelnames=self.get_labels_for_metric(
|
|
"litellm_deployment_success_responses"
|
|
),
|
|
)
|
|
self.litellm_deployment_failure_responses = self._counter_factory(
|
|
name="litellm_deployment_failure_responses",
|
|
documentation="LLM Deployment Analytics - Total number of failed LLM API calls for a specific LLM deploymeny. exception_status is the status of the exception from the llm api",
|
|
labelnames=self.get_labels_for_metric(
|
|
"litellm_deployment_failure_responses"
|
|
),
|
|
)
|
|
|
|
self.litellm_deployment_total_requests = self._counter_factory(
|
|
name="litellm_deployment_total_requests",
|
|
documentation="LLM Deployment Analytics - Total number of LLM API calls via litellm - success + failure",
|
|
labelnames=self.get_labels_for_metric(
|
|
"litellm_deployment_total_requests"
|
|
),
|
|
)
|
|
|
|
# Deployment Latency tracking
|
|
self.litellm_deployment_latency_per_output_token = self._histogram_factory(
|
|
name="litellm_deployment_latency_per_output_token",
|
|
documentation="LLM Deployment Analytics - Latency per output token",
|
|
labelnames=self.get_labels_for_metric(
|
|
"litellm_deployment_latency_per_output_token"
|
|
),
|
|
)
|
|
|
|
self.litellm_deployment_successful_fallbacks = self._counter_factory(
|
|
"litellm_deployment_successful_fallbacks",
|
|
"LLM Deployment Analytics - Number of successful fallback requests from primary model -> fallback model",
|
|
self.get_labels_for_metric("litellm_deployment_successful_fallbacks"),
|
|
)
|
|
|
|
self.litellm_deployment_failed_fallbacks = self._counter_factory(
|
|
"litellm_deployment_failed_fallbacks",
|
|
"LLM Deployment Analytics - Number of failed fallback requests from primary model -> fallback model",
|
|
self.get_labels_for_metric("litellm_deployment_failed_fallbacks"),
|
|
)
|
|
|
|
# Callback Logging Failure Metrics
|
|
self.litellm_callback_logging_failures_metric = self._counter_factory(
|
|
name="litellm_callback_logging_failures_metric",
|
|
documentation="Total number of failures when emitting logs to callbacks (e.g. s3_v2, langfuse, etc)",
|
|
labelnames=["callback_name"],
|
|
)
|
|
|
|
self.litellm_llm_api_failed_requests_metric = self._counter_factory(
|
|
name="litellm_llm_api_failed_requests_metric",
|
|
documentation="deprecated - use litellm_proxy_failed_requests_metric",
|
|
labelnames=self.get_labels_for_metric(
|
|
"litellm_llm_api_failed_requests_metric"
|
|
),
|
|
)
|
|
|
|
self.litellm_requests_metric = self._counter_factory(
|
|
name="litellm_requests_metric",
|
|
documentation="deprecated - use litellm_proxy_total_requests_metric. Total number of LLM calls to litellm - track total per API Key, team, user",
|
|
labelnames=self.get_labels_for_metric("litellm_requests_metric"),
|
|
)
|
|
|
|
# Cache metrics
|
|
self.litellm_cache_hits_metric = self._counter_factory(
|
|
name="litellm_cache_hits_metric",
|
|
documentation="Total number of LiteLLM cache hits",
|
|
labelnames=self.get_labels_for_metric("litellm_cache_hits_metric"),
|
|
)
|
|
|
|
self.litellm_cache_misses_metric = self._counter_factory(
|
|
name="litellm_cache_misses_metric",
|
|
documentation="Total number of LiteLLM cache misses",
|
|
labelnames=self.get_labels_for_metric("litellm_cache_misses_metric"),
|
|
)
|
|
|
|
self.litellm_cached_tokens_metric = self._counter_factory(
|
|
name="litellm_cached_tokens_metric",
|
|
documentation="Total tokens served from LiteLLM cache",
|
|
labelnames=self.get_labels_for_metric("litellm_cached_tokens_metric"),
|
|
)
|
|
|
|
# User and Team count metrics
|
|
self.litellm_total_users_metric = self._gauge_factory(
|
|
"litellm_total_users",
|
|
"Total number of users in LiteLLM",
|
|
labelnames=[],
|
|
)
|
|
|
|
self.litellm_teams_count_metric = self._gauge_factory(
|
|
"litellm_teams_count",
|
|
"Total number of teams in LiteLLM",
|
|
labelnames=[],
|
|
)
|
|
|
|
########################################
|
|
# Managed Batch Metrics
|
|
########################################
|
|
self.litellm_managed_batch_created_total = self._counter_factory(
|
|
name="litellm_managed_batch_created_total",
|
|
documentation="Total number of managed batches created",
|
|
labelnames=[
|
|
"model",
|
|
"api_provider",
|
|
"user",
|
|
"user_email",
|
|
"api_key_alias",
|
|
],
|
|
)
|
|
|
|
self.litellm_managed_file_size_bytes = self._gauge_factory(
|
|
"litellm_managed_file_size_bytes",
|
|
"Size of the most recent managed batch file in bytes (last-seen value per label combination)",
|
|
labelnames=["purpose", "file_type", "model", "api_provider", "user"],
|
|
)
|
|
|
|
self.litellm_managed_batch_duration_seconds = self._histogram_factory(
|
|
"litellm_managed_batch_duration_seconds",
|
|
"Duration of completed managed batches in seconds (completed_at - created_at)",
|
|
labelnames=["model", "api_provider"],
|
|
buckets=BATCH_DURATION_BUCKETS,
|
|
)
|
|
|
|
self.litellm_managed_file_created_total = self._counter_factory(
|
|
name="litellm_managed_file_created_total",
|
|
documentation="Total number of managed files created",
|
|
labelnames=[
|
|
"model",
|
|
"api_provider",
|
|
"user",
|
|
"user_email",
|
|
"api_key_alias",
|
|
],
|
|
)
|
|
|
|
self.litellm_managed_file_deleted_total = self._counter_factory(
|
|
name="litellm_managed_file_deleted_total",
|
|
documentation="Total number of managed file deletions (success or blocked)",
|
|
labelnames=["result"],
|
|
)
|
|
|
|
self.litellm_check_batch_cost_jobs_polled = self._gauge_factory(
|
|
"litellm_check_batch_cost_jobs_polled",
|
|
"Number of unprocessed batches found by the last CheckBatchCost poll",
|
|
labelnames=[],
|
|
)
|
|
|
|
self.litellm_check_batch_cost_jobs_processed_total = self._counter_factory(
|
|
name="litellm_check_batch_cost_jobs_processed_total",
|
|
documentation="Total number of batches successfully cost-tracked by CheckBatchCost",
|
|
labelnames=["model", "api_provider"],
|
|
)
|
|
|
|
self.litellm_check_batch_cost_errors_total = self._counter_factory(
|
|
name="litellm_check_batch_cost_errors_total",
|
|
documentation="Total number of errors in CheckBatchCost by error type",
|
|
labelnames=["error_type"],
|
|
)
|
|
|
|
self.litellm_check_batch_cost_last_run_timestamp = self._gauge_factory(
|
|
"litellm_check_batch_cost_last_run_timestamp",
|
|
"Unix timestamp of the last CheckBatchCost job run",
|
|
labelnames=[],
|
|
)
|
|
|
|
except Exception as e:
|
|
print_verbose(f"Got exception on init prometheus client {str(e)}")
|
|
raise e
|
|
|
|
def _parse_prometheus_config(self) -> Dict[str, List[str]]:
|
|
"""Parse prometheus metrics configuration for label filtering and enabled metrics"""
|
|
import litellm
|
|
from litellm.types.integrations.prometheus import PrometheusMetricsConfig
|
|
|
|
config = litellm.prometheus_metrics_config
|
|
|
|
# If no config is provided, return empty dict (no filtering)
|
|
if not config:
|
|
return {}
|
|
|
|
verbose_logger.debug(f"prometheus config: {config}")
|
|
|
|
# Parse and validate all configuration groups
|
|
parsed_configs = []
|
|
self.enabled_metrics = set()
|
|
|
|
for group_config in config:
|
|
if isinstance(group_config, dict):
|
|
parsed_config = PrometheusMetricsConfig(**group_config)
|
|
else:
|
|
parsed_config = group_config
|
|
|
|
parsed_configs.append(parsed_config)
|
|
self.enabled_metrics.update(parsed_config.metrics)
|
|
|
|
# Validate all configurations
|
|
validation_results = self._validate_all_configurations(parsed_configs)
|
|
|
|
if validation_results.has_errors:
|
|
self._pretty_print_validation_errors(validation_results)
|
|
error_message = "Configuration validation failed:\n" + "\n".join(
|
|
validation_results.all_error_messages
|
|
)
|
|
raise ValueError(error_message)
|
|
|
|
# Build label filters from valid configurations
|
|
label_filters = self._build_label_filters(parsed_configs)
|
|
|
|
# Pretty print the processed configuration
|
|
self._pretty_print_prometheus_config(label_filters)
|
|
return label_filters
|
|
|
|
def _validate_all_configurations(self, parsed_configs: List) -> ValidationResults:
|
|
"""Validate all metric configurations and return collected errors"""
|
|
metric_errors = []
|
|
label_errors = []
|
|
|
|
for config in parsed_configs:
|
|
for metric_name in config.metrics:
|
|
# Validate metric name
|
|
metric_error = self._validate_single_metric_name(metric_name)
|
|
if metric_error:
|
|
metric_errors.append(metric_error)
|
|
continue # Skip label validation if metric name is invalid
|
|
|
|
# Validate labels if provided
|
|
if config.include_labels:
|
|
label_error = self._validate_single_metric_labels(
|
|
metric_name, config.include_labels
|
|
)
|
|
if label_error:
|
|
label_errors.append(label_error)
|
|
|
|
return ValidationResults(metric_errors=metric_errors, label_errors=label_errors)
|
|
|
|
def _validate_single_metric_name(
|
|
self, metric_name: str
|
|
) -> Optional[MetricValidationError]:
|
|
"""Validate a single metric name"""
|
|
from typing import get_args
|
|
|
|
if metric_name not in set(get_args(DEFINED_PROMETHEUS_METRICS)):
|
|
return MetricValidationError(
|
|
metric_name=metric_name,
|
|
valid_metrics=get_args(DEFINED_PROMETHEUS_METRICS),
|
|
)
|
|
return None
|
|
|
|
def _validate_single_metric_labels(
|
|
self, metric_name: str, labels: List[str]
|
|
) -> Optional[LabelValidationError]:
|
|
"""Validate labels for a single metric"""
|
|
from typing import cast
|
|
|
|
# Get valid labels for this metric from PrometheusMetricLabels
|
|
valid_labels = PrometheusMetricLabels.get_labels(
|
|
cast(DEFINED_PROMETHEUS_METRICS, metric_name)
|
|
)
|
|
|
|
# Find invalid labels
|
|
invalid_labels = [label for label in labels if label not in valid_labels]
|
|
|
|
if invalid_labels:
|
|
return LabelValidationError(
|
|
metric_name=metric_name,
|
|
invalid_labels=invalid_labels,
|
|
valid_labels=valid_labels,
|
|
)
|
|
return None
|
|
|
|
def _build_label_filters(self, parsed_configs: List) -> Dict[str, List[str]]:
|
|
"""Build label filters from validated configurations"""
|
|
label_filters = {}
|
|
|
|
for config in parsed_configs:
|
|
for metric_name in config.metrics:
|
|
if config.include_labels:
|
|
# Only add if metric name is valid (validation already passed)
|
|
if self._validate_single_metric_name(metric_name) is None:
|
|
label_filters[metric_name] = config.include_labels
|
|
|
|
return label_filters
|
|
|
|
def _validate_configured_metric_labels(self, metric_name: str, labels: List[str]):
|
|
"""
|
|
Ensure that all the configured labels are valid for the metric
|
|
|
|
Raises ValueError if the metric labels are invalid and pretty prints the error
|
|
"""
|
|
label_error = self._validate_single_metric_labels(metric_name, labels)
|
|
if label_error:
|
|
self._pretty_print_invalid_labels_error(
|
|
metric_name=label_error.metric_name,
|
|
invalid_labels=label_error.invalid_labels,
|
|
valid_labels=label_error.valid_labels,
|
|
)
|
|
raise ValueError(label_error.message)
|
|
|
|
return True
|
|
|
|
#########################################################
|
|
# Pretty print functions
|
|
#########################################################
|
|
|
|
def _pretty_print_validation_errors(
|
|
self, validation_results: ValidationResults
|
|
) -> None:
|
|
"""Pretty print all validation errors using rich"""
|
|
try:
|
|
from rich.console import Console
|
|
from rich.panel import Panel
|
|
from rich.table import Table
|
|
from rich.text import Text
|
|
|
|
console = Console()
|
|
|
|
# Create error panel title
|
|
title = Text("🚨🚨 Configuration Validation Errors", style="bold red")
|
|
|
|
# Print main error panel
|
|
console.print("\n")
|
|
console.print(Panel(title, border_style="red"))
|
|
|
|
# Show invalid metric names if any
|
|
if validation_results.metric_errors:
|
|
invalid_metrics = [
|
|
e.metric_name for e in validation_results.metric_errors
|
|
]
|
|
valid_metrics = validation_results.metric_errors[
|
|
0
|
|
].valid_metrics # All should have same valid metrics
|
|
|
|
metrics_error_text = Text(
|
|
f"Invalid Metric Names: {', '.join(invalid_metrics)}",
|
|
style="bold red",
|
|
)
|
|
console.print(Panel(metrics_error_text, border_style="red"))
|
|
|
|
metrics_table = Table(
|
|
title="📊 Valid Metric Names",
|
|
show_header=True,
|
|
header_style="bold green",
|
|
title_justify="left",
|
|
border_style="green",
|
|
)
|
|
metrics_table.add_column(
|
|
"Available Metrics", style="cyan", no_wrap=True
|
|
)
|
|
|
|
for metric in sorted(valid_metrics):
|
|
metrics_table.add_row(metric)
|
|
|
|
console.print(metrics_table)
|
|
|
|
# Show invalid labels if any
|
|
if validation_results.label_errors:
|
|
for error in validation_results.label_errors:
|
|
labels_error_text = Text(
|
|
f"Invalid Labels for '{error.metric_name}': {', '.join(error.invalid_labels)}",
|
|
style="bold red",
|
|
)
|
|
console.print(Panel(labels_error_text, border_style="red"))
|
|
|
|
labels_table = Table(
|
|
title=f"🏷️ Valid Labels for '{error.metric_name}'",
|
|
show_header=True,
|
|
header_style="bold green",
|
|
title_justify="left",
|
|
border_style="green",
|
|
)
|
|
labels_table.add_column("Valid Labels", style="cyan", no_wrap=True)
|
|
|
|
for label in sorted(error.valid_labels):
|
|
labels_table.add_row(label)
|
|
|
|
console.print(labels_table)
|
|
|
|
console.print("\n")
|
|
|
|
except ImportError:
|
|
# Fallback to simple logging if rich is not available
|
|
for metric_error in validation_results.metric_errors:
|
|
verbose_logger.error(metric_error.message)
|
|
for label_error in validation_results.label_errors:
|
|
verbose_logger.error(label_error.message)
|
|
|
|
def _pretty_print_invalid_labels_error(
|
|
self, metric_name: str, invalid_labels: List[str], valid_labels: List[str]
|
|
) -> None:
|
|
"""Pretty print error message for invalid labels using rich"""
|
|
try:
|
|
from rich.console import Console
|
|
from rich.panel import Panel
|
|
from rich.table import Table
|
|
from rich.text import Text
|
|
|
|
console = Console()
|
|
|
|
# Create error panel title
|
|
title = Text(
|
|
f"🚨🚨 Invalid Labels for Metric: '{metric_name}'\nInvalid labels: {', '.join(invalid_labels)}\nPlease specify only valid labels below",
|
|
style="bold red",
|
|
)
|
|
|
|
# Create valid labels table
|
|
labels_table = Table(
|
|
title="🏷️ Valid Labels for this Metric",
|
|
show_header=True,
|
|
header_style="bold green",
|
|
title_justify="left",
|
|
border_style="green",
|
|
)
|
|
labels_table.add_column("Valid Labels", style="cyan", no_wrap=True)
|
|
|
|
for label in sorted(valid_labels):
|
|
labels_table.add_row(label)
|
|
|
|
# Print everything in a nice panel
|
|
console.print("\n")
|
|
console.print(Panel(title, border_style="red"))
|
|
console.print(labels_table)
|
|
console.print("\n")
|
|
|
|
except ImportError:
|
|
# Fallback to simple logging if rich is not available
|
|
verbose_logger.error(
|
|
f"Invalid labels for metric '{metric_name}': {invalid_labels}. Valid labels: {sorted(valid_labels)}"
|
|
)
|
|
|
|
def _pretty_print_invalid_metric_error(
|
|
self, invalid_metric_name: str, valid_metrics: tuple
|
|
) -> None:
|
|
"""Pretty print error message for invalid metric name using rich"""
|
|
try:
|
|
from rich.console import Console
|
|
from rich.panel import Panel
|
|
from rich.table import Table
|
|
from rich.text import Text
|
|
|
|
console = Console()
|
|
|
|
# Create error panel title
|
|
title = Text(
|
|
f"🚨🚨 Invalid Metric Name: '{invalid_metric_name}'\nPlease specify one of the allowed metrics below",
|
|
style="bold red",
|
|
)
|
|
|
|
# Create valid metrics table
|
|
metrics_table = Table(
|
|
title="📊 Valid Metric Names",
|
|
show_header=True,
|
|
header_style="bold green",
|
|
title_justify="left",
|
|
border_style="green",
|
|
)
|
|
metrics_table.add_column("Available Metrics", style="cyan", no_wrap=True)
|
|
|
|
for metric in sorted(valid_metrics):
|
|
metrics_table.add_row(metric)
|
|
|
|
# Print everything in a nice panel
|
|
console.print("\n")
|
|
console.print(Panel(title, border_style="red"))
|
|
console.print(metrics_table)
|
|
console.print("\n")
|
|
|
|
except ImportError:
|
|
# Fallback to simple logging if rich is not available
|
|
verbose_logger.error(
|
|
f"Invalid metric name: {invalid_metric_name}. Valid metrics: {sorted(valid_metrics)}"
|
|
)
|
|
|
|
#########################################################
|
|
# End of pretty print functions
|
|
#########################################################
|
|
|
|
def _valid_metric_name(self, metric_name: str):
|
|
"""
|
|
Raises ValueError if the metric name is invalid and pretty prints the error
|
|
"""
|
|
error = self._validate_single_metric_name(metric_name)
|
|
if error:
|
|
self._pretty_print_invalid_metric_error(
|
|
invalid_metric_name=error.metric_name, valid_metrics=error.valid_metrics
|
|
)
|
|
raise ValueError(error.message)
|
|
|
|
def _pretty_print_prometheus_config(
|
|
self, label_filters: Dict[str, List[str]]
|
|
) -> None:
|
|
"""Pretty print the processed prometheus configuration using rich"""
|
|
try:
|
|
from rich.console import Console
|
|
from rich.panel import Panel
|
|
from rich.table import Table
|
|
from rich.text import Text
|
|
|
|
console = Console()
|
|
|
|
# Create main panel title
|
|
title = Text("Prometheus Configuration Processed", style="bold blue")
|
|
|
|
# Create enabled metrics table
|
|
metrics_table = Table(
|
|
title="📊 Enabled Metrics",
|
|
show_header=True,
|
|
header_style="bold magenta",
|
|
title_justify="left",
|
|
)
|
|
metrics_table.add_column("Metric Name", style="cyan", no_wrap=True)
|
|
|
|
if hasattr(self, "enabled_metrics") and self.enabled_metrics:
|
|
for metric in sorted(self.enabled_metrics):
|
|
metrics_table.add_row(metric)
|
|
else:
|
|
metrics_table.add_row(
|
|
"[yellow]All metrics enabled (no filter applied)[/yellow]"
|
|
)
|
|
|
|
# Create label filters table
|
|
labels_table = Table(
|
|
title="🏷️ Label Filters",
|
|
show_header=True,
|
|
header_style="bold green",
|
|
title_justify="left",
|
|
)
|
|
labels_table.add_column("Metric Name", style="cyan", no_wrap=True)
|
|
labels_table.add_column("Allowed Labels", style="yellow")
|
|
|
|
if label_filters:
|
|
for metric_name, labels in sorted(label_filters.items()):
|
|
labels_str = (
|
|
", ".join(labels)
|
|
if labels
|
|
else "[dim]No labels specified[/dim]"
|
|
)
|
|
labels_table.add_row(metric_name, labels_str)
|
|
else:
|
|
labels_table.add_row(
|
|
"[yellow]No label filtering applied[/yellow]",
|
|
"[dim]All default labels will be used[/dim]",
|
|
)
|
|
|
|
# Print everything in a nice panel
|
|
console.print("\n")
|
|
console.print(Panel(title, border_style="blue"))
|
|
console.print(metrics_table)
|
|
console.print(labels_table)
|
|
console.print("\n")
|
|
|
|
except ImportError:
|
|
# Fallback to simple logging if rich is not available
|
|
verbose_logger.info(
|
|
f"Enabled metrics: {sorted(self.enabled_metrics) if hasattr(self, 'enabled_metrics') else 'All metrics'}"
|
|
)
|
|
verbose_logger.info(f"Label filters: {label_filters}")
|
|
|
|
def _is_metric_enabled(self, metric_name: str) -> bool:
|
|
"""Check if a metric is enabled based on configuration"""
|
|
# If no specific configuration is provided, enable all metrics (default behavior)
|
|
if not hasattr(self, "enabled_metrics"):
|
|
return True
|
|
|
|
# If enabled_metrics is empty, enable all metrics
|
|
if not self.enabled_metrics:
|
|
return True
|
|
|
|
return metric_name in self.enabled_metrics
|
|
|
|
def _create_metric_factory(self, metric_class):
|
|
"""Create a factory function that returns either a real metric or a no-op metric"""
|
|
|
|
def factory(*args, **kwargs):
|
|
# Extract metric name from the first argument or 'name' keyword argument
|
|
metric_name = args[0] if args else kwargs.get("name", "")
|
|
|
|
if self._is_metric_enabled(metric_name):
|
|
return metric_class(*args, **kwargs)
|
|
else:
|
|
return NoOpMetric()
|
|
|
|
return factory
|
|
|
|
def get_labels_for_metric(
|
|
self, metric_name: DEFINED_PROMETHEUS_METRICS
|
|
) -> List[str]:
|
|
"""
|
|
Get the labels for a metric, filtered if configured
|
|
"""
|
|
# Get default labels for this metric from PrometheusMetricLabels
|
|
default_labels = PrometheusMetricLabels.get_labels(metric_name)
|
|
|
|
# If no label filtering is configured for this metric, use default labels
|
|
if metric_name not in self.label_filters:
|
|
return default_labels
|
|
|
|
# Get configured labels for this metric
|
|
configured_labels = self.label_filters[metric_name]
|
|
|
|
# Return intersection of configured and default labels to ensure we only use valid labels
|
|
filtered_labels = [
|
|
label for label in default_labels if label in configured_labels
|
|
]
|
|
|
|
return filtered_labels
|
|
|
|
def _inc_labeled_counter(
|
|
self,
|
|
counter: Any,
|
|
metric_name: DEFINED_PROMETHEUS_METRICS,
|
|
enum_values: UserAPIKeyLabelValues,
|
|
label_context: Optional[PrometheusLabelFactoryContext] = None,
|
|
amount: float = 1.0,
|
|
) -> None:
|
|
_labels = prometheus_label_factory(
|
|
supported_enum_labels=self.get_labels_for_metric(metric_name=metric_name),
|
|
enum_values=enum_values,
|
|
label_context=label_context,
|
|
)
|
|
counter.labels(**_labels).inc(amount)
|
|
|
|
async def async_log_success_event(self, kwargs, response_obj, start_time, end_time):
|
|
# Define prometheus client
|
|
verbose_logger.debug(
|
|
"prometheus Logging - Enters success logging function (kwargs keys: %s)",
|
|
list(kwargs.keys()) if isinstance(kwargs, dict) else type(kwargs).__name__,
|
|
)
|
|
|
|
# unpack kwargs
|
|
standard_logging_payload: Optional[StandardLoggingPayload] = kwargs.get(
|
|
"standard_logging_object"
|
|
)
|
|
|
|
if standard_logging_payload is None or not isinstance(
|
|
standard_logging_payload, dict
|
|
):
|
|
raise ValueError(
|
|
f"standard_logging_object is required, got={standard_logging_payload}"
|
|
)
|
|
|
|
if self._should_skip_metrics_for_invalid_key(
|
|
kwargs=kwargs, standard_logging_payload=standard_logging_payload
|
|
):
|
|
return
|
|
|
|
model = kwargs.get("model", "")
|
|
litellm_params = kwargs.get("litellm_params", {}) or {}
|
|
_metadata = litellm_params.get("metadata") or {}
|
|
get_end_user_id_for_cost_tracking = _get_cached_end_user_id_for_cost_tracking()
|
|
|
|
end_user_id = get_end_user_id_for_cost_tracking(
|
|
litellm_params, service_type="prometheus"
|
|
)
|
|
user_id = standard_logging_payload["metadata"]["user_api_key_user_id"]
|
|
user_api_key = standard_logging_payload["metadata"]["user_api_key_hash"]
|
|
user_api_key_alias = standard_logging_payload["metadata"]["user_api_key_alias"]
|
|
user_api_team = standard_logging_payload["metadata"]["user_api_key_team_id"]
|
|
user_api_team_alias = standard_logging_payload["metadata"][
|
|
"user_api_key_team_alias"
|
|
]
|
|
user_api_key_org_id = standard_logging_payload["metadata"].get(
|
|
"user_api_key_org_id"
|
|
)
|
|
user_api_key_org_alias = standard_logging_payload["metadata"].get(
|
|
"user_api_key_org_alias"
|
|
)
|
|
output_tokens = standard_logging_payload["completion_tokens"]
|
|
tokens_used = standard_logging_payload["total_tokens"]
|
|
response_cost = standard_logging_payload["response_cost"]
|
|
_requester_metadata: Optional[dict] = standard_logging_payload["metadata"].get(
|
|
"requester_metadata"
|
|
)
|
|
user_api_key_auth_metadata: Optional[dict] = standard_logging_payload[
|
|
"metadata"
|
|
].get("user_api_key_auth_metadata")
|
|
spend_logs_metadata: Optional[dict] = standard_logging_payload["metadata"].get(
|
|
"spend_logs_metadata"
|
|
)
|
|
|
|
combined_metadata: Dict[str, Any] = {
|
|
**(_requester_metadata if _requester_metadata else {}),
|
|
**(user_api_key_auth_metadata if user_api_key_auth_metadata else {}),
|
|
**(spend_logs_metadata if spend_logs_metadata else {}),
|
|
}
|
|
if standard_logging_payload is not None and isinstance(
|
|
standard_logging_payload, dict
|
|
):
|
|
_tags = standard_logging_payload["request_tags"]
|
|
else:
|
|
_tags = []
|
|
|
|
print_verbose(
|
|
f"inside track_prometheus_metrics, model {model}, response_cost {response_cost}, tokens_used {tokens_used}, end_user_id {end_user_id}, user_api_key {user_api_key}"
|
|
)
|
|
|
|
enum_values = UserAPIKeyLabelValues(
|
|
end_user=end_user_id,
|
|
hashed_api_key=user_api_key,
|
|
api_key_alias=user_api_key_alias,
|
|
requested_model=standard_logging_payload["model_group"],
|
|
model_group=standard_logging_payload["model_group"],
|
|
team=user_api_team,
|
|
team_alias=user_api_team_alias,
|
|
org_id=user_api_key_org_id,
|
|
org_alias=user_api_key_org_alias,
|
|
user=user_id,
|
|
user_email=standard_logging_payload["metadata"]["user_api_key_user_email"],
|
|
status_code="200",
|
|
model=model,
|
|
litellm_model_name=model,
|
|
tags=_tags,
|
|
model_id=standard_logging_payload["model_id"],
|
|
api_base=standard_logging_payload["api_base"],
|
|
api_provider=standard_logging_payload["custom_llm_provider"],
|
|
exception_status=None,
|
|
exception_class=None,
|
|
custom_metadata_labels=get_custom_labels_from_metadata(
|
|
metadata=combined_metadata
|
|
),
|
|
route=standard_logging_payload["metadata"].get(
|
|
"user_api_key_request_route"
|
|
),
|
|
client_ip=standard_logging_payload["metadata"].get("requester_ip_address"),
|
|
user_agent=standard_logging_payload["metadata"].get("user_agent"),
|
|
stream=(
|
|
str(standard_logging_payload.get("stream"))
|
|
if litellm.prometheus_emit_stream_label
|
|
else None
|
|
),
|
|
)
|
|
|
|
if (
|
|
user_api_key is not None
|
|
and isinstance(user_api_key, str)
|
|
and user_api_key.startswith("sk-")
|
|
):
|
|
from litellm.proxy.utils import hash_token
|
|
|
|
user_api_key = hash_token(user_api_key)
|
|
|
|
label_context = PrometheusLabelFactoryContext(
|
|
enum_values
|
|
) # amortized per request.
|
|
|
|
# increment total LLM requests and spend metric
|
|
self._increment_top_level_request_and_spend_metrics(
|
|
end_user_id=end_user_id,
|
|
user_api_key=user_api_key,
|
|
user_api_key_alias=user_api_key_alias,
|
|
model=model,
|
|
user_api_team=user_api_team,
|
|
user_api_team_alias=user_api_team_alias,
|
|
user_id=user_id,
|
|
response_cost=response_cost,
|
|
enum_values=enum_values,
|
|
label_context=label_context,
|
|
)
|
|
|
|
# input, output, total token metrics
|
|
self._increment_token_metrics(
|
|
# why type ignore below?
|
|
# 1. We just checked if isinstance(standard_logging_payload, dict). Pyright complains.
|
|
# 2. Pyright does not allow us to run isinstance(standard_logging_payload, StandardLoggingPayload) <- this would be ideal
|
|
standard_logging_payload=standard_logging_payload, # type: ignore
|
|
end_user_id=end_user_id,
|
|
user_api_key=user_api_key,
|
|
user_api_key_alias=user_api_key_alias,
|
|
model=model,
|
|
user_api_team=user_api_team,
|
|
user_api_team_alias=user_api_team_alias,
|
|
user_id=user_id,
|
|
enum_values=enum_values,
|
|
label_context=label_context,
|
|
)
|
|
|
|
# remaining budget metrics
|
|
await self._increment_remaining_budget_metrics(
|
|
user_api_team=user_api_team,
|
|
user_api_team_alias=user_api_team_alias,
|
|
user_api_key=user_api_key,
|
|
user_api_key_alias=user_api_key_alias,
|
|
litellm_params=litellm_params,
|
|
response_cost=response_cost,
|
|
user_id=user_id,
|
|
user_api_key_org_id=user_api_key_org_id,
|
|
)
|
|
|
|
# set proxy virtual key rpm/tpm metrics
|
|
self._set_virtual_key_rate_limit_metrics(
|
|
user_api_key=user_api_key,
|
|
user_api_key_alias=user_api_key_alias,
|
|
kwargs=kwargs,
|
|
metadata=_metadata,
|
|
model_id=enum_values.model_id,
|
|
)
|
|
|
|
# set latency metrics
|
|
self._set_latency_metrics(
|
|
kwargs=kwargs,
|
|
model=model,
|
|
user_api_key=user_api_key,
|
|
user_api_key_alias=user_api_key_alias,
|
|
user_api_team=user_api_team,
|
|
user_api_team_alias=user_api_team_alias,
|
|
# why type ignore below?
|
|
# 1. We just checked if isinstance(standard_logging_payload, dict). Pyright complains.
|
|
# 2. Pyright does not allow us to run isinstance(standard_logging_payload, StandardLoggingPayload) <- this would be ideal
|
|
enum_values=enum_values,
|
|
label_context=label_context,
|
|
)
|
|
|
|
# set x-ratelimit headers
|
|
self.set_llm_deployment_success_metrics(
|
|
kwargs,
|
|
start_time,
|
|
end_time,
|
|
enum_values,
|
|
output_tokens,
|
|
label_context=label_context,
|
|
)
|
|
|
|
# cache metrics
|
|
self._increment_cache_metrics(
|
|
standard_logging_payload=standard_logging_payload, # type: ignore
|
|
enum_values=enum_values,
|
|
label_context=label_context,
|
|
)
|
|
|
|
# increment litellm_proxy_total_requests_metric for all successful requests
|
|
# (both streaming and non-streaming) in this single location to prevent
|
|
# double-counting that occurs when async_post_call_success_hook also increments
|
|
PrometheusLogger._inc_labeled_counter(
|
|
self,
|
|
self.litellm_proxy_total_requests_metric,
|
|
"litellm_proxy_total_requests_metric",
|
|
enum_values,
|
|
label_context=label_context,
|
|
)
|
|
|
|
def _increment_token_metrics(
|
|
self,
|
|
standard_logging_payload: StandardLoggingPayload,
|
|
end_user_id: Optional[str],
|
|
user_api_key: Optional[str],
|
|
user_api_key_alias: Optional[str],
|
|
model: Optional[str],
|
|
user_api_team: Optional[str],
|
|
user_api_team_alias: Optional[str],
|
|
user_id: Optional[str],
|
|
enum_values: UserAPIKeyLabelValues,
|
|
label_context: Optional[PrometheusLabelFactoryContext] = None,
|
|
):
|
|
verbose_logger.debug("prometheus Logging - Enters token metrics function")
|
|
# token metrics
|
|
|
|
if standard_logging_payload is not None and isinstance(
|
|
standard_logging_payload, dict
|
|
):
|
|
_tags = standard_logging_payload["request_tags"]
|
|
|
|
PrometheusLogger._inc_labeled_counter(
|
|
self,
|
|
self.litellm_tokens_metric,
|
|
"litellm_total_tokens_metric",
|
|
enum_values,
|
|
label_context=label_context,
|
|
amount=float(standard_logging_payload["total_tokens"]),
|
|
)
|
|
PrometheusLogger._inc_labeled_counter(
|
|
self,
|
|
self.litellm_input_tokens_metric,
|
|
"litellm_input_tokens_metric",
|
|
enum_values,
|
|
label_context=label_context,
|
|
amount=float(standard_logging_payload["prompt_tokens"]),
|
|
)
|
|
PrometheusLogger._inc_labeled_counter(
|
|
self,
|
|
self.litellm_output_tokens_metric,
|
|
"litellm_output_tokens_metric",
|
|
enum_values,
|
|
label_context=label_context,
|
|
amount=float(standard_logging_payload["completion_tokens"]),
|
|
)
|
|
|
|
def _increment_cache_metrics(
|
|
self,
|
|
standard_logging_payload: StandardLoggingPayload,
|
|
enum_values: UserAPIKeyLabelValues,
|
|
label_context: Optional[PrometheusLabelFactoryContext] = None,
|
|
):
|
|
"""
|
|
Increment cache-related Prometheus metrics based on cache hit/miss status.
|
|
|
|
Args:
|
|
standard_logging_payload: Contains cache_hit field (True/False/None)
|
|
enum_values: Label values for Prometheus metrics
|
|
"""
|
|
cache_hit = standard_logging_payload.get("cache_hit")
|
|
|
|
# Only track if cache_hit has a definite value (True or False)
|
|
if cache_hit is None:
|
|
return
|
|
|
|
if cache_hit is True:
|
|
# Increment cache hits counter
|
|
PrometheusLogger._inc_labeled_counter(
|
|
self,
|
|
self.litellm_cache_hits_metric,
|
|
"litellm_cache_hits_metric",
|
|
enum_values,
|
|
label_context=label_context,
|
|
)
|
|
|
|
# Increment cached tokens counter
|
|
total_tokens = standard_logging_payload.get("total_tokens", 0)
|
|
if total_tokens > 0:
|
|
PrometheusLogger._inc_labeled_counter(
|
|
self,
|
|
self.litellm_cached_tokens_metric,
|
|
"litellm_cached_tokens_metric",
|
|
enum_values,
|
|
label_context=label_context,
|
|
amount=float(total_tokens),
|
|
)
|
|
else:
|
|
# cache_hit is False - increment cache misses counter
|
|
PrometheusLogger._inc_labeled_counter(
|
|
self,
|
|
self.litellm_cache_misses_metric,
|
|
"litellm_cache_misses_metric",
|
|
enum_values,
|
|
label_context=label_context,
|
|
)
|
|
|
|
async def _increment_remaining_budget_metrics(
|
|
self,
|
|
user_api_team: Optional[str],
|
|
user_api_team_alias: Optional[str],
|
|
user_api_key: Optional[str],
|
|
user_api_key_alias: Optional[str],
|
|
litellm_params: dict,
|
|
response_cost: float,
|
|
user_id: Optional[str] = None,
|
|
user_api_key_org_id: Optional[str] = None,
|
|
):
|
|
_metadata = litellm_params.get("metadata") or {}
|
|
_team_spend = _metadata.get("user_api_key_team_spend", None)
|
|
_team_max_budget = _metadata.get("user_api_key_team_max_budget", None)
|
|
|
|
_api_key_spend = _metadata.get("user_api_key_spend", None)
|
|
_api_key_max_budget = _metadata.get("user_api_key_max_budget", None)
|
|
|
|
_user_spend = _metadata.get("user_api_key_user_spend", None)
|
|
_user_max_budget = _metadata.get("user_api_key_user_max_budget", None)
|
|
|
|
results = await asyncio.gather(
|
|
self._set_api_key_budget_metrics_after_api_request(
|
|
user_api_key=user_api_key,
|
|
user_api_key_alias=user_api_key_alias,
|
|
response_cost=response_cost,
|
|
key_max_budget=_api_key_max_budget,
|
|
key_spend=_api_key_spend,
|
|
),
|
|
self._set_team_budget_metrics_after_api_request(
|
|
user_api_team=user_api_team,
|
|
user_api_team_alias=user_api_team_alias,
|
|
team_spend=_team_spend,
|
|
team_max_budget=_team_max_budget,
|
|
response_cost=response_cost,
|
|
),
|
|
self._set_user_budget_metrics_after_api_request(
|
|
user_id=user_id,
|
|
user_spend=_user_spend,
|
|
user_max_budget=_user_max_budget,
|
|
response_cost=response_cost,
|
|
),
|
|
self._set_org_budget_metrics_after_api_request(
|
|
org_id=user_api_key_org_id,
|
|
response_cost=response_cost,
|
|
),
|
|
return_exceptions=True,
|
|
)
|
|
for i, r in enumerate(results):
|
|
if isinstance(r, Exception):
|
|
verbose_logger.debug(
|
|
f"[Non-Blocking] Prometheus: Budget metric lookup {['key', 'team', 'user', 'org'][i]} failed: {r}"
|
|
)
|
|
|
|
def _increment_top_level_request_and_spend_metrics(
|
|
self,
|
|
end_user_id: Optional[str],
|
|
user_api_key: Optional[str],
|
|
user_api_key_alias: Optional[str],
|
|
model: Optional[str],
|
|
user_api_team: Optional[str],
|
|
user_api_team_alias: Optional[str],
|
|
user_id: Optional[str],
|
|
response_cost: float,
|
|
enum_values: UserAPIKeyLabelValues,
|
|
label_context: Optional[PrometheusLabelFactoryContext] = None,
|
|
):
|
|
PrometheusLogger._inc_labeled_counter(
|
|
self,
|
|
self.litellm_requests_metric,
|
|
"litellm_requests_metric",
|
|
enum_values,
|
|
label_context=label_context,
|
|
)
|
|
PrometheusLogger._inc_labeled_counter(
|
|
self,
|
|
self.litellm_spend_metric,
|
|
"litellm_spend_metric",
|
|
enum_values,
|
|
label_context=label_context,
|
|
amount=float(response_cost),
|
|
)
|
|
|
|
def _set_virtual_key_rate_limit_metrics(
|
|
self,
|
|
user_api_key: Optional[str],
|
|
user_api_key_alias: Optional[str],
|
|
kwargs: dict,
|
|
metadata: dict,
|
|
model_id: Optional[str] = None,
|
|
):
|
|
from litellm.proxy.common_utils.callback_utils import (
|
|
get_model_group_from_litellm_kwargs,
|
|
)
|
|
|
|
# Set remaining rpm/tpm for API Key + model
|
|
# see parallel_request_limiter.py - variables are set there
|
|
model_group = get_model_group_from_litellm_kwargs(kwargs)
|
|
remaining_requests_variable_name = (
|
|
f"litellm-key-remaining-requests-{model_group}"
|
|
)
|
|
remaining_tokens_variable_name = f"litellm-key-remaining-tokens-{model_group}"
|
|
|
|
remaining_requests = (
|
|
metadata.get(remaining_requests_variable_name, sys.maxsize) or sys.maxsize
|
|
)
|
|
remaining_tokens = (
|
|
metadata.get(remaining_tokens_variable_name, sys.maxsize) or sys.maxsize
|
|
)
|
|
|
|
self.litellm_remaining_api_key_requests_for_model.labels(
|
|
_sanitize_prometheus_label_value(user_api_key),
|
|
_sanitize_prometheus_label_value(user_api_key_alias),
|
|
_sanitize_prometheus_label_value(model_group),
|
|
_sanitize_prometheus_label_value(model_id),
|
|
).set(remaining_requests)
|
|
|
|
self.litellm_remaining_api_key_tokens_for_model.labels(
|
|
_sanitize_prometheus_label_value(user_api_key),
|
|
_sanitize_prometheus_label_value(user_api_key_alias),
|
|
_sanitize_prometheus_label_value(model_group),
|
|
_sanitize_prometheus_label_value(model_id),
|
|
).set(remaining_tokens)
|
|
|
|
def _set_latency_metrics(
|
|
self,
|
|
kwargs: dict,
|
|
model: Optional[str],
|
|
user_api_key: Optional[str],
|
|
user_api_key_alias: Optional[str],
|
|
user_api_team: Optional[str],
|
|
user_api_team_alias: Optional[str],
|
|
enum_values: UserAPIKeyLabelValues,
|
|
label_context: Optional[PrometheusLabelFactoryContext] = None,
|
|
):
|
|
# latency metrics
|
|
end_time: datetime = kwargs.get("end_time") or datetime.now()
|
|
start_time: Optional[datetime] = kwargs.get("start_time")
|
|
api_call_start_time = kwargs.get("api_call_start_time", None)
|
|
completion_start_time = kwargs.get("completion_start_time", None)
|
|
time_to_first_token_seconds = self._safe_duration_seconds(
|
|
start_time=api_call_start_time,
|
|
end_time=completion_start_time,
|
|
)
|
|
if (
|
|
time_to_first_token_seconds is not None
|
|
and kwargs.get("stream", False) is True # only emit for streaming requests
|
|
):
|
|
_ttft_labels = prometheus_label_factory(
|
|
supported_enum_labels=self.get_labels_for_metric(
|
|
metric_name="litellm_llm_api_time_to_first_token_metric"
|
|
),
|
|
enum_values=enum_values,
|
|
label_context=label_context,
|
|
)
|
|
self.litellm_llm_api_time_to_first_token_metric.labels(
|
|
**_ttft_labels
|
|
).observe(time_to_first_token_seconds)
|
|
else:
|
|
verbose_logger.debug(
|
|
"Time to first token metric not emitted, stream option in model_parameters is not True"
|
|
)
|
|
|
|
api_call_total_time_seconds = self._safe_duration_seconds(
|
|
start_time=api_call_start_time,
|
|
end_time=end_time,
|
|
)
|
|
if api_call_total_time_seconds is not None:
|
|
_labels = prometheus_label_factory(
|
|
supported_enum_labels=self.get_labels_for_metric(
|
|
metric_name="litellm_llm_api_latency_metric"
|
|
),
|
|
enum_values=enum_values,
|
|
label_context=label_context,
|
|
)
|
|
self.litellm_llm_api_latency_metric.labels(**_labels).observe(
|
|
api_call_total_time_seconds
|
|
)
|
|
|
|
# total request latency
|
|
total_time_seconds = self._safe_duration_seconds(
|
|
start_time=start_time,
|
|
end_time=end_time,
|
|
)
|
|
if total_time_seconds is not None:
|
|
_labels = prometheus_label_factory(
|
|
supported_enum_labels=self.get_labels_for_metric(
|
|
metric_name="litellm_request_total_latency_metric"
|
|
),
|
|
enum_values=enum_values,
|
|
label_context=label_context,
|
|
)
|
|
self.litellm_request_total_latency_metric.labels(**_labels).observe(
|
|
total_time_seconds
|
|
)
|
|
|
|
# request queue time (time from arrival to processing start)
|
|
_litellm_params = kwargs.get("litellm_params", {}) or {}
|
|
queue_time_seconds = (_litellm_params.get("metadata") or {}).get(
|
|
"queue_time_seconds"
|
|
)
|
|
if queue_time_seconds is not None and queue_time_seconds >= 0:
|
|
_labels = prometheus_label_factory(
|
|
supported_enum_labels=self.get_labels_for_metric(
|
|
metric_name="litellm_request_queue_time_seconds"
|
|
),
|
|
enum_values=enum_values,
|
|
label_context=label_context,
|
|
)
|
|
self.litellm_request_queue_time_metric.labels(**_labels).observe(
|
|
queue_time_seconds
|
|
)
|
|
|
|
async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time):
|
|
verbose_logger.debug(
|
|
"prometheus Logging - Enters failure logging function (kwargs keys: %s)",
|
|
list(kwargs.keys()) if isinstance(kwargs, dict) else type(kwargs).__name__,
|
|
)
|
|
|
|
standard_logging_payload: StandardLoggingPayload = kwargs.get(
|
|
"standard_logging_object", {}
|
|
)
|
|
|
|
if self._should_skip_metrics_for_invalid_key(
|
|
kwargs=kwargs, standard_logging_payload=standard_logging_payload
|
|
):
|
|
return
|
|
|
|
model = kwargs.get("model", "")
|
|
|
|
litellm_params = kwargs.get("litellm_params", {}) or {}
|
|
get_end_user_id_for_cost_tracking = _get_cached_end_user_id_for_cost_tracking()
|
|
|
|
end_user_id = get_end_user_id_for_cost_tracking(
|
|
litellm_params, service_type="prometheus"
|
|
)
|
|
user_id = standard_logging_payload["metadata"]["user_api_key_user_id"]
|
|
user_api_key = standard_logging_payload["metadata"]["user_api_key_hash"]
|
|
user_api_key_alias = standard_logging_payload["metadata"]["user_api_key_alias"]
|
|
user_api_team = standard_logging_payload["metadata"]["user_api_key_team_id"]
|
|
user_api_team_alias = standard_logging_payload["metadata"][
|
|
"user_api_key_team_alias"
|
|
]
|
|
user_api_key_org_id = standard_logging_payload["metadata"].get(
|
|
"user_api_key_org_id"
|
|
)
|
|
|
|
try:
|
|
self.litellm_llm_api_failed_requests_metric.labels(
|
|
_sanitize_prometheus_label_value(end_user_id),
|
|
_sanitize_prometheus_label_value(user_api_key),
|
|
_sanitize_prometheus_label_value(user_api_key_alias),
|
|
_sanitize_prometheus_label_value(model),
|
|
_sanitize_prometheus_label_value(user_api_team),
|
|
_sanitize_prometheus_label_value(user_api_team_alias),
|
|
_sanitize_prometheus_label_value(user_id),
|
|
_sanitize_prometheus_label_value(
|
|
standard_logging_payload.get("model_id", "")
|
|
),
|
|
).inc()
|
|
self.set_llm_deployment_failure_metrics(kwargs)
|
|
await self._set_org_budget_metrics_after_api_request(
|
|
org_id=user_api_key_org_id,
|
|
response_cost=0,
|
|
)
|
|
except Exception as e:
|
|
verbose_logger.exception(
|
|
"prometheus Layer Error(): Exception occured - {}".format(str(e))
|
|
)
|
|
pass
|
|
pass
|
|
|
|
def _extract_status_code(
|
|
self,
|
|
kwargs: Optional[dict] = None,
|
|
enum_values: Optional[Any] = None,
|
|
exception: Optional[Exception] = None,
|
|
) -> Optional[int]:
|
|
"""
|
|
Extract HTTP status code from various input formats for validation.
|
|
|
|
This is a centralized helper to extract status code from different
|
|
callback function signatures. Handles both ProxyException (uses 'code')
|
|
and standard exceptions (uses 'status_code').
|
|
|
|
Args:
|
|
kwargs: Dictionary potentially containing 'exception' key
|
|
enum_values: Object with 'status_code' attribute
|
|
exception: Exception object to extract status code from directly
|
|
|
|
Returns:
|
|
Status code as integer if found, None otherwise
|
|
"""
|
|
status_code = None
|
|
|
|
# Try from enum_values first (most common in our callbacks)
|
|
if (
|
|
enum_values
|
|
and hasattr(enum_values, "status_code")
|
|
and enum_values.status_code
|
|
):
|
|
try:
|
|
status_code = int(enum_values.status_code)
|
|
except (ValueError, TypeError):
|
|
pass
|
|
|
|
if not status_code and exception:
|
|
# ProxyException uses 'code' attribute, other exceptions may use 'status_code'
|
|
status_code = getattr(exception, "status_code", None) or getattr(
|
|
exception, "code", None
|
|
)
|
|
if status_code is not None:
|
|
try:
|
|
status_code = int(status_code)
|
|
except (ValueError, TypeError):
|
|
status_code = None
|
|
|
|
if not status_code and kwargs:
|
|
exception_in_kwargs = kwargs.get("exception")
|
|
if exception_in_kwargs:
|
|
status_code = getattr(
|
|
exception_in_kwargs, "status_code", None
|
|
) or getattr(exception_in_kwargs, "code", None)
|
|
if status_code is not None:
|
|
try:
|
|
status_code = int(status_code)
|
|
except (ValueError, TypeError):
|
|
status_code = None
|
|
|
|
return status_code
|
|
|
|
def _is_invalid_api_key_request(
|
|
self,
|
|
status_code: Optional[int],
|
|
exception: Optional[Exception] = None,
|
|
) -> bool:
|
|
"""
|
|
Determine if a request has an invalid API key based on status code and exception.
|
|
|
|
This method prevents invalid authentication attempts from being recorded in
|
|
Prometheus metrics. A 401 status code is the definitive indicator of authentication
|
|
failure. Additionally, we check exception messages for authentication error patterns
|
|
to catch cases where the exception hasn't been converted to a ProxyException yet.
|
|
|
|
Args:
|
|
status_code: HTTP status code (401 indicates authentication error)
|
|
exception: Exception object to check for auth-related error messages
|
|
|
|
Returns:
|
|
True if the request has an invalid API key and metrics should be skipped,
|
|
False otherwise
|
|
"""
|
|
if status_code == 401:
|
|
return True
|
|
|
|
# Handle cases where AssertionError is raised before conversion to ProxyException
|
|
if exception is not None:
|
|
exception_str = str(exception).lower()
|
|
auth_error_patterns = [
|
|
"virtual key expected",
|
|
"expected to start with 'sk-'",
|
|
"authentication error",
|
|
"invalid api key",
|
|
"api key not valid",
|
|
]
|
|
if any(pattern in exception_str for pattern in auth_error_patterns):
|
|
return True
|
|
|
|
return False
|
|
|
|
def _should_skip_metrics_for_invalid_key(
|
|
self,
|
|
kwargs: Optional[dict] = None,
|
|
user_api_key_dict: Optional[Any] = None,
|
|
enum_values: Optional[Any] = None,
|
|
standard_logging_payload: Optional[Union[dict, StandardLoggingPayload]] = None,
|
|
exception: Optional[Exception] = None,
|
|
) -> bool:
|
|
"""
|
|
Determine if Prometheus metrics should be skipped for invalid API key requests.
|
|
|
|
This is a centralized validation method that extracts status code and exception
|
|
information from various callback function signatures and determines if the request
|
|
represents an invalid API key attempt that should be filtered from metrics.
|
|
|
|
Args:
|
|
kwargs: Dictionary potentially containing exception and other data
|
|
user_api_key_dict: User API key authentication object (currently unused)
|
|
enum_values: Object with status_code attribute
|
|
standard_logging_payload: Standard logging payload dictionary
|
|
exception: Exception object to check directly
|
|
|
|
Returns:
|
|
True if metrics should be skipped (invalid key detected), False otherwise
|
|
"""
|
|
status_code = self._extract_status_code(
|
|
kwargs=kwargs,
|
|
enum_values=enum_values,
|
|
exception=exception,
|
|
)
|
|
|
|
if exception is None and kwargs:
|
|
exception = kwargs.get("exception")
|
|
|
|
if self._is_invalid_api_key_request(status_code, exception=exception):
|
|
verbose_logger.debug(
|
|
"Skipping Prometheus metrics for invalid API key request: "
|
|
f"status_code={status_code}, exception={type(exception).__name__ if exception else None}"
|
|
)
|
|
return True
|
|
|
|
return False
|
|
|
|
async def async_post_call_failure_hook(
|
|
self,
|
|
request_data: dict,
|
|
original_exception: Exception,
|
|
user_api_key_dict: UserAPIKeyAuth,
|
|
traceback_str: Optional[str] = None,
|
|
):
|
|
"""
|
|
Track client side failures
|
|
|
|
Proxy level tracking - failed client side requests
|
|
|
|
labelnames=[
|
|
"end_user",
|
|
"hashed_api_key",
|
|
"api_key_alias",
|
|
REQUESTED_MODEL,
|
|
"team",
|
|
"team_alias",
|
|
] + EXCEPTION_LABELS,
|
|
"""
|
|
from litellm.litellm_core_utils.litellm_logging import (
|
|
StandardLoggingPayloadSetup,
|
|
)
|
|
|
|
if self._should_skip_metrics_for_invalid_key(
|
|
user_api_key_dict=user_api_key_dict,
|
|
exception=original_exception,
|
|
):
|
|
return
|
|
|
|
status_code = self._extract_status_code(exception=original_exception)
|
|
|
|
try:
|
|
_tags = StandardLoggingPayloadSetup._get_request_tags(
|
|
litellm_params=request_data,
|
|
proxy_server_request=request_data.get("proxy_server_request", {}),
|
|
)
|
|
_metadata = request_data.get("metadata", {}) or {}
|
|
model_id = _metadata.get("model_info", {}).get("id") or request_data.get(
|
|
"model_info", {}
|
|
).get("id")
|
|
enum_values = UserAPIKeyLabelValues(
|
|
end_user=user_api_key_dict.end_user_id,
|
|
user=user_api_key_dict.user_id,
|
|
user_email=user_api_key_dict.user_email,
|
|
hashed_api_key=user_api_key_dict.api_key,
|
|
api_key_alias=user_api_key_dict.key_alias,
|
|
team=user_api_key_dict.team_id,
|
|
team_alias=user_api_key_dict.team_alias,
|
|
org_id=user_api_key_dict.org_id,
|
|
org_alias=user_api_key_dict.organization_alias,
|
|
requested_model=request_data.get("model", ""),
|
|
status_code=str(status_code),
|
|
exception_status=str(status_code),
|
|
exception_class=self._get_exception_class_name(original_exception),
|
|
tags=_tags,
|
|
route=user_api_key_dict.request_route,
|
|
client_ip=_metadata.get("requester_ip_address"),
|
|
user_agent=_metadata.get("user_agent"),
|
|
model_id=model_id,
|
|
stream=(
|
|
str(request_data.get("stream"))
|
|
if litellm.prometheus_emit_stream_label
|
|
else None
|
|
),
|
|
)
|
|
_label_ctx = PrometheusLabelFactoryContext(enum_values)
|
|
PrometheusLogger._inc_labeled_counter(
|
|
self,
|
|
self.litellm_proxy_failed_requests_metric,
|
|
"litellm_proxy_failed_requests_metric",
|
|
enum_values,
|
|
label_context=_label_ctx,
|
|
)
|
|
PrometheusLogger._inc_labeled_counter(
|
|
self,
|
|
self.litellm_proxy_total_requests_metric,
|
|
"litellm_proxy_total_requests_metric",
|
|
enum_values,
|
|
label_context=_label_ctx,
|
|
)
|
|
|
|
except Exception as e:
|
|
verbose_logger.exception(
|
|
"prometheus Layer Error(): Exception occured - {}".format(str(e))
|
|
)
|
|
pass
|
|
|
|
async def async_post_call_success_hook(
|
|
self, data: dict, user_api_key_dict: UserAPIKeyAuth, response
|
|
):
|
|
"""
|
|
Proxy level tracking - triggered when the proxy responds with a success response to the client
|
|
|
|
Note: litellm_proxy_total_requests_metric is NOT incremented here to avoid
|
|
double-counting. It is incremented in async_log_success_event which fires
|
|
for all successful requests (both streaming and non-streaming).
|
|
"""
|
|
pass
|
|
|
|
def _safe_get(self, obj: Any, key: str, default: Any = None) -> Any:
|
|
"""Get value from dict or Pydantic model."""
|
|
if obj is None:
|
|
return default
|
|
if isinstance(obj, dict):
|
|
return obj.get(key, default)
|
|
return getattr(obj, key, default)
|
|
|
|
def _extract_deployment_failure_label_values(
|
|
self, request_kwargs: dict
|
|
) -> Dict[str, Optional[str]]:
|
|
"""
|
|
Extract label values for deployment failure metrics from all available
|
|
sources in request_kwargs. Falls back to litellm_params metadata and
|
|
user_api_key_auth when standard_logging_payload has None values.
|
|
"""
|
|
standard_logging_payload = (
|
|
request_kwargs.get("standard_logging_object", {}) or {}
|
|
)
|
|
_litellm_params = request_kwargs.get("litellm_params", {}) or {}
|
|
_metadata_raw = self._safe_get(standard_logging_payload, "metadata") or {}
|
|
if isinstance(_metadata_raw, dict):
|
|
_metadata = _metadata_raw
|
|
else:
|
|
_metadata = {
|
|
"user_api_key_alias": getattr(
|
|
_metadata_raw, "user_api_key_alias", None
|
|
),
|
|
"user_api_key_team_id": getattr(
|
|
_metadata_raw, "user_api_key_team_id", None
|
|
),
|
|
"user_api_key_team_alias": getattr(
|
|
_metadata_raw, "user_api_key_team_alias", None
|
|
),
|
|
"user_api_key_hash": getattr(_metadata_raw, "user_api_key_hash", None),
|
|
"requester_ip_address": getattr(
|
|
_metadata_raw, "requester_ip_address", None
|
|
),
|
|
"user_agent": getattr(_metadata_raw, "user_agent", None),
|
|
}
|
|
_litellm_params_metadata = _litellm_params.get("metadata", {}) or {}
|
|
|
|
# Extract user_api_key_auth if present (proxy injects this, skipped in merge)
|
|
user_api_key_auth = _litellm_params_metadata.get("user_api_key_auth")
|
|
|
|
def _get_api_key_alias() -> Optional[str]:
|
|
val = _metadata.get("user_api_key_alias")
|
|
if val is not None:
|
|
return val
|
|
val = _litellm_params_metadata.get("user_api_key_alias")
|
|
if val is not None:
|
|
return val
|
|
if user_api_key_auth is not None:
|
|
return getattr(user_api_key_auth, "key_alias", None)
|
|
return None
|
|
|
|
def _get_team_id() -> Optional[str]:
|
|
val = _metadata.get("user_api_key_team_id")
|
|
if val is not None:
|
|
return val
|
|
val = _litellm_params_metadata.get("user_api_key_team_id")
|
|
if val is not None:
|
|
return val
|
|
if user_api_key_auth is not None:
|
|
return getattr(user_api_key_auth, "team_id", None)
|
|
return None
|
|
|
|
def _get_team_alias() -> Optional[str]:
|
|
val = _metadata.get("user_api_key_team_alias")
|
|
if val is not None:
|
|
return val
|
|
val = _litellm_params_metadata.get("user_api_key_team_alias")
|
|
if val is not None:
|
|
return val
|
|
if user_api_key_auth is not None:
|
|
return getattr(user_api_key_auth, "team_alias", None)
|
|
return None
|
|
|
|
def _get_hashed_api_key() -> Optional[str]:
|
|
val = _metadata.get("user_api_key_hash")
|
|
if val is not None:
|
|
return val
|
|
val = _litellm_params_metadata.get("user_api_key_hash")
|
|
if val is not None:
|
|
return val
|
|
if user_api_key_auth is not None:
|
|
return getattr(user_api_key_auth, "api_key", None) or getattr(
|
|
user_api_key_auth, "api_key_hash", None
|
|
)
|
|
return None
|
|
|
|
return {
|
|
"api_key_alias": _get_api_key_alias(),
|
|
"team": _get_team_id(),
|
|
"team_alias": _get_team_alias(),
|
|
"hashed_api_key": _get_hashed_api_key(),
|
|
"client_ip": _metadata.get("requester_ip_address")
|
|
or _litellm_params_metadata.get("requester_ip_address"),
|
|
"user_agent": _metadata.get("user_agent")
|
|
or _litellm_params_metadata.get("user_agent"),
|
|
}
|
|
|
|
def set_llm_deployment_failure_metrics(self, request_kwargs: dict):
|
|
"""
|
|
Sets Failure metrics when an LLM API call fails
|
|
|
|
- mark the deployment as partial outage
|
|
- increment deployment failure responses metric
|
|
- increment deployment total requests metric
|
|
|
|
Args:
|
|
request_kwargs: dict
|
|
|
|
"""
|
|
try:
|
|
verbose_logger.debug("setting remaining tokens requests metric")
|
|
standard_logging_payload: StandardLoggingPayload = request_kwargs.get(
|
|
"standard_logging_object", {}
|
|
)
|
|
_litellm_params = request_kwargs.get("litellm_params", {}) or {}
|
|
litellm_model_name = request_kwargs.get("model", None)
|
|
model_group = standard_logging_payload.get("model_group", None)
|
|
api_base = standard_logging_payload.get("api_base", None)
|
|
model_id = standard_logging_payload.get("model_id", None)
|
|
exception = request_kwargs.get("exception", None)
|
|
|
|
# Fallback: model_id from litellm_metadata.model_info
|
|
if model_id is None:
|
|
_model_info = (
|
|
(_litellm_params.get("litellm_metadata") or {}).get("model_info")
|
|
or (_litellm_params.get("metadata") or {}).get("model_info")
|
|
or {}
|
|
)
|
|
model_id = _model_info.get("id")
|
|
|
|
# Fallback: model_group from litellm_metadata
|
|
if model_group is None:
|
|
model_group = (_litellm_params.get("litellm_metadata") or {}).get(
|
|
"model_group"
|
|
) or (_litellm_params.get("metadata") or {}).get("model_group")
|
|
|
|
llm_provider = _litellm_params.get("custom_llm_provider", None)
|
|
|
|
if self._should_skip_metrics_for_invalid_key(
|
|
kwargs=request_kwargs,
|
|
standard_logging_payload=standard_logging_payload,
|
|
):
|
|
return
|
|
|
|
# Extract context labels from all available sources (fix for None labels)
|
|
fallback_values = self._extract_deployment_failure_label_values(
|
|
request_kwargs
|
|
)
|
|
_metadata = standard_logging_payload.get("metadata", {}) or {}
|
|
hashed_api_key = fallback_values.get("hashed_api_key") or _metadata.get(
|
|
"user_api_key_hash"
|
|
)
|
|
api_key_alias = fallback_values.get("api_key_alias") or _metadata.get(
|
|
"user_api_key_alias"
|
|
)
|
|
team = fallback_values.get("team") or _metadata.get("user_api_key_team_id")
|
|
team_alias = fallback_values.get("team_alias") or _metadata.get(
|
|
"user_api_key_team_alias"
|
|
)
|
|
client_ip = fallback_values.get("client_ip") or _metadata.get(
|
|
"requester_ip_address"
|
|
)
|
|
user_agent = fallback_values.get("user_agent") or _metadata.get(
|
|
"user_agent"
|
|
)
|
|
|
|
# exception_status: prefer status_code, fallback to exception class for known types
|
|
exception_status = None
|
|
if exception is not None:
|
|
exception_status = str(getattr(exception, "status_code", None))
|
|
if exception_status == "None" or not exception_status:
|
|
code = getattr(exception, "code", None)
|
|
if code is not None:
|
|
exception_status = str(code)
|
|
|
|
# Create enum_values for the label factory (always create for use in different metrics)
|
|
enum_values = UserAPIKeyLabelValues(
|
|
litellm_model_name=litellm_model_name,
|
|
model_id=model_id,
|
|
api_base=api_base,
|
|
api_provider=llm_provider,
|
|
exception_status=exception_status,
|
|
exception_class=(
|
|
self._get_exception_class_name(exception) if exception else None
|
|
),
|
|
requested_model=model_group or litellm_model_name,
|
|
hashed_api_key=hashed_api_key,
|
|
api_key_alias=api_key_alias,
|
|
team=team,
|
|
team_alias=team_alias,
|
|
tags=standard_logging_payload.get("request_tags", []),
|
|
client_ip=client_ip,
|
|
user_agent=user_agent,
|
|
)
|
|
|
|
"""
|
|
log these labels
|
|
["litellm_model_name", "model_id", "api_base", "api_provider"]
|
|
"""
|
|
self.set_deployment_partial_outage(
|
|
litellm_model_name=litellm_model_name or "",
|
|
model_id=model_id,
|
|
api_base=api_base,
|
|
api_provider=llm_provider or "",
|
|
)
|
|
_deployment_label_ctx = PrometheusLabelFactoryContext(enum_values)
|
|
if exception is not None:
|
|
PrometheusLogger._inc_labeled_counter(
|
|
self,
|
|
self.litellm_deployment_failure_responses,
|
|
"litellm_deployment_failure_responses",
|
|
enum_values,
|
|
label_context=_deployment_label_ctx,
|
|
)
|
|
|
|
PrometheusLogger._inc_labeled_counter(
|
|
self,
|
|
self.litellm_deployment_total_requests,
|
|
"litellm_deployment_total_requests",
|
|
enum_values,
|
|
label_context=_deployment_label_ctx,
|
|
)
|
|
|
|
pass
|
|
except Exception as e:
|
|
verbose_logger.debug(
|
|
"Prometheus Error: set_llm_deployment_failure_metrics. Exception occured - {}".format(
|
|
str(e)
|
|
)
|
|
)
|
|
|
|
def _set_deployment_tpm_rpm_limit_metrics(
|
|
self,
|
|
model_info: dict,
|
|
litellm_params: dict,
|
|
litellm_model_name: Optional[str],
|
|
model_id: Optional[str],
|
|
api_base: Optional[str],
|
|
llm_provider: Optional[str],
|
|
):
|
|
"""
|
|
Set the deployment TPM and RPM limits metrics
|
|
"""
|
|
tpm = model_info.get("tpm") or litellm_params.get("tpm")
|
|
rpm = model_info.get("rpm") or litellm_params.get("rpm")
|
|
|
|
if tpm is not None:
|
|
_labels = prometheus_label_factory(
|
|
supported_enum_labels=self.get_labels_for_metric(
|
|
metric_name="litellm_deployment_tpm_limit"
|
|
),
|
|
enum_values=UserAPIKeyLabelValues(
|
|
litellm_model_name=litellm_model_name,
|
|
model_id=model_id,
|
|
api_base=api_base,
|
|
api_provider=llm_provider,
|
|
),
|
|
)
|
|
self.litellm_deployment_tpm_limit.labels(**_labels).set(tpm)
|
|
|
|
if rpm is not None:
|
|
_labels = prometheus_label_factory(
|
|
supported_enum_labels=self.get_labels_for_metric(
|
|
metric_name="litellm_deployment_rpm_limit"
|
|
),
|
|
enum_values=UserAPIKeyLabelValues(
|
|
litellm_model_name=litellm_model_name,
|
|
model_id=model_id,
|
|
api_base=api_base,
|
|
api_provider=llm_provider,
|
|
),
|
|
)
|
|
self.litellm_deployment_rpm_limit.labels(**_labels).set(rpm)
|
|
|
|
def set_llm_deployment_success_metrics(
|
|
self,
|
|
request_kwargs: dict,
|
|
start_time,
|
|
end_time,
|
|
enum_values: UserAPIKeyLabelValues,
|
|
output_tokens: float = 1.0,
|
|
label_context: Optional[PrometheusLabelFactoryContext] = None,
|
|
):
|
|
try:
|
|
verbose_logger.debug("setting remaining tokens requests metric")
|
|
standard_logging_payload: Optional[StandardLoggingPayload] = (
|
|
request_kwargs.get("standard_logging_object")
|
|
)
|
|
|
|
if standard_logging_payload is None:
|
|
return
|
|
|
|
# Skip recording metrics for invalid API key requests
|
|
if self._should_skip_metrics_for_invalid_key(
|
|
kwargs=request_kwargs,
|
|
enum_values=enum_values,
|
|
standard_logging_payload=standard_logging_payload,
|
|
):
|
|
return
|
|
|
|
api_base = standard_logging_payload["api_base"]
|
|
_litellm_params = request_kwargs.get("litellm_params", {}) or {}
|
|
_metadata = get_litellm_metadata_from_kwargs(request_kwargs)
|
|
litellm_model_name = request_kwargs.get("model", None)
|
|
llm_provider = _litellm_params.get("custom_llm_provider", None)
|
|
_model_info = _metadata.get("model_info") or {}
|
|
model_id = _model_info.get("id", None)
|
|
|
|
if _model_info or _litellm_params:
|
|
self._set_deployment_tpm_rpm_limit_metrics(
|
|
model_info=_model_info,
|
|
litellm_params=_litellm_params,
|
|
litellm_model_name=litellm_model_name,
|
|
model_id=model_id,
|
|
api_base=api_base,
|
|
llm_provider=llm_provider,
|
|
)
|
|
|
|
remaining_requests: Optional[int] = None
|
|
remaining_tokens: Optional[int] = None
|
|
if additional_headers := standard_logging_payload["hidden_params"][
|
|
"additional_headers"
|
|
]:
|
|
# OpenAI / OpenAI Compatible headers
|
|
remaining_requests = additional_headers.get(
|
|
"x_ratelimit_remaining_requests", None
|
|
)
|
|
remaining_tokens = additional_headers.get(
|
|
"x_ratelimit_remaining_tokens", None
|
|
)
|
|
|
|
if litellm_overhead_time_ms := standard_logging_payload[
|
|
"hidden_params"
|
|
].get("litellm_overhead_time_ms"):
|
|
_labels = prometheus_label_factory(
|
|
supported_enum_labels=self.get_labels_for_metric(
|
|
metric_name="litellm_overhead_latency_metric"
|
|
),
|
|
enum_values=enum_values,
|
|
label_context=label_context,
|
|
)
|
|
self.litellm_overhead_latency_metric.labels(**_labels).observe(
|
|
litellm_overhead_time_ms / 1000
|
|
) # set as seconds
|
|
|
|
if remaining_requests:
|
|
"""
|
|
"model_group",
|
|
"api_provider",
|
|
"api_base",
|
|
"litellm_model_name"
|
|
"""
|
|
_labels = prometheus_label_factory(
|
|
supported_enum_labels=self.get_labels_for_metric(
|
|
metric_name="litellm_remaining_requests_metric"
|
|
),
|
|
enum_values=enum_values,
|
|
label_context=label_context,
|
|
)
|
|
self.litellm_remaining_requests_metric.labels(**_labels).set(
|
|
remaining_requests
|
|
)
|
|
|
|
if remaining_tokens:
|
|
_labels = prometheus_label_factory(
|
|
supported_enum_labels=self.get_labels_for_metric(
|
|
metric_name="litellm_remaining_tokens_metric"
|
|
),
|
|
enum_values=enum_values,
|
|
label_context=label_context,
|
|
)
|
|
self.litellm_remaining_tokens_metric.labels(**_labels).set(
|
|
remaining_tokens
|
|
)
|
|
|
|
"""
|
|
log these labels
|
|
["litellm_model_name", "requested_model", model_id", "api_base", "api_provider"]
|
|
"""
|
|
self.set_deployment_healthy(
|
|
litellm_model_name=litellm_model_name or "",
|
|
model_id=model_id or "",
|
|
api_base=api_base or "",
|
|
api_provider=llm_provider or "",
|
|
)
|
|
|
|
PrometheusLogger._inc_labeled_counter(
|
|
self,
|
|
self.litellm_deployment_success_responses,
|
|
"litellm_deployment_success_responses",
|
|
enum_values,
|
|
label_context=label_context,
|
|
)
|
|
PrometheusLogger._inc_labeled_counter(
|
|
self,
|
|
self.litellm_deployment_total_requests,
|
|
"litellm_deployment_total_requests",
|
|
enum_values,
|
|
label_context=label_context,
|
|
)
|
|
|
|
# Track deployment Latency
|
|
response_ms: timedelta = end_time - start_time
|
|
time_to_first_token_response_time: Optional[timedelta] = None
|
|
|
|
if (
|
|
request_kwargs.get("stream", None) is not None
|
|
and request_kwargs["stream"] is True
|
|
):
|
|
# only log ttft for streaming request
|
|
time_to_first_token_response_time = (
|
|
request_kwargs.get("completion_start_time", end_time) - start_time
|
|
)
|
|
|
|
# use the metric that is not None
|
|
# if streaming - use time_to_first_token_response
|
|
# if not streaming - use response_ms
|
|
_latency: timedelta = time_to_first_token_response_time or response_ms
|
|
_latency_seconds = _latency.total_seconds()
|
|
|
|
# latency per output token
|
|
latency_per_token = None
|
|
if output_tokens is not None and output_tokens > 0:
|
|
latency_per_token = _latency_seconds / output_tokens
|
|
_labels = prometheus_label_factory(
|
|
supported_enum_labels=self.get_labels_for_metric(
|
|
metric_name="litellm_deployment_latency_per_output_token"
|
|
),
|
|
enum_values=enum_values,
|
|
label_context=label_context,
|
|
)
|
|
self.litellm_deployment_latency_per_output_token.labels(
|
|
**_labels
|
|
).observe(latency_per_token)
|
|
|
|
except Exception as e:
|
|
verbose_logger.exception(
|
|
"Prometheus Error: set_llm_deployment_success_metrics. Exception occured - {}".format(
|
|
str(e)
|
|
)
|
|
)
|
|
return
|
|
|
|
def _record_guardrail_metrics(
|
|
self,
|
|
guardrail_name: str,
|
|
latency_seconds: float,
|
|
status: str,
|
|
error_type: Optional[str],
|
|
hook_type: str,
|
|
):
|
|
"""
|
|
Record guardrail metrics for prometheus.
|
|
|
|
Args:
|
|
guardrail_name: Name of the guardrail
|
|
latency_seconds: Execution latency in seconds
|
|
status: "success" or "error"
|
|
error_type: Type of error if any, None otherwise
|
|
hook_type: "pre_call", "during_call", or "post_call"
|
|
"""
|
|
try:
|
|
# Record latency
|
|
self.litellm_guardrail_latency_metric.labels(
|
|
guardrail_name=guardrail_name,
|
|
status=status,
|
|
error_type=error_type or "none",
|
|
hook_type=hook_type,
|
|
).observe(latency_seconds)
|
|
|
|
# Record request count
|
|
self.litellm_guardrail_requests_total.labels(
|
|
guardrail_name=guardrail_name,
|
|
status=status,
|
|
hook_type=hook_type,
|
|
).inc()
|
|
|
|
# Record error count if there was an error
|
|
if status == "error" and error_type:
|
|
self.litellm_guardrail_errors_total.labels(
|
|
guardrail_name=guardrail_name,
|
|
error_type=error_type,
|
|
hook_type=hook_type,
|
|
).inc()
|
|
except Exception as e:
|
|
verbose_logger.debug(f"Error recording guardrail metrics: {str(e)}")
|
|
|
|
########################################
|
|
# Managed Batch Metric Recording Methods
|
|
########################################
|
|
|
|
def record_managed_batch_created(
|
|
self,
|
|
model: Optional[str],
|
|
api_provider: Optional[str],
|
|
user: Optional[str],
|
|
user_email: Optional[str],
|
|
api_key_alias: Optional[str],
|
|
):
|
|
try:
|
|
self.litellm_managed_batch_created_total.labels(
|
|
model=model,
|
|
api_provider=api_provider,
|
|
user=user,
|
|
user_email=user_email,
|
|
api_key_alias=api_key_alias,
|
|
).inc()
|
|
except Exception as e:
|
|
verbose_logger.warning(f"Error recording batch created metric: {e}")
|
|
|
|
def record_managed_file_size(
|
|
self,
|
|
size_bytes: int,
|
|
purpose: str,
|
|
file_type: str,
|
|
model: Optional[str] = None,
|
|
api_provider: Optional[str] = None,
|
|
user: Optional[str] = None,
|
|
):
|
|
"""Record the size of a managed file. Uses a gauge (last-seen value per label combination)."""
|
|
try:
|
|
self.litellm_managed_file_size_bytes.labels(
|
|
purpose=purpose,
|
|
file_type=file_type,
|
|
model=model or "",
|
|
api_provider=api_provider or "",
|
|
user=user or "",
|
|
).set(size_bytes)
|
|
except Exception as e:
|
|
verbose_logger.warning(f"Error recording file size metric: {e}")
|
|
|
|
def record_managed_batch_duration(
|
|
self,
|
|
duration_seconds: float,
|
|
model: Optional[str] = None,
|
|
api_provider: Optional[str] = None,
|
|
):
|
|
try:
|
|
self.litellm_managed_batch_duration_seconds.labels(
|
|
model=model or "",
|
|
api_provider=api_provider or "",
|
|
).observe(duration_seconds)
|
|
except Exception as e:
|
|
verbose_logger.warning(f"Error recording batch duration metric: {e}")
|
|
|
|
def record_managed_file_created(
|
|
self,
|
|
model: Optional[str],
|
|
api_provider: Optional[str],
|
|
user: Optional[str],
|
|
user_email: Optional[str],
|
|
api_key_alias: Optional[str],
|
|
):
|
|
try:
|
|
self.litellm_managed_file_created_total.labels(
|
|
model=model,
|
|
api_provider=api_provider,
|
|
user=user,
|
|
user_email=user_email,
|
|
api_key_alias=api_key_alias,
|
|
).inc()
|
|
except Exception as e:
|
|
verbose_logger.warning(f"Error recording file created metric: {e}")
|
|
|
|
def record_managed_file_deleted(self, result: str):
|
|
"""Record a managed file deletion attempt. result is 'success' or 'blocked'."""
|
|
try:
|
|
self.litellm_managed_file_deleted_total.labels(result=result).inc()
|
|
except Exception as e:
|
|
verbose_logger.warning(f"Error recording file deleted metric: {e}")
|
|
|
|
def record_check_batch_cost_run(
|
|
self,
|
|
jobs_polled: int,
|
|
processed_models: Optional[List[Tuple[Optional[str], Optional[str]]]] = None,
|
|
):
|
|
"""
|
|
Record CheckBatchCost polling metrics.
|
|
|
|
Args:
|
|
jobs_polled: Number of unprocessed batches found
|
|
processed_models: List of (model, api_provider) tuples for processed jobs
|
|
"""
|
|
import time
|
|
|
|
try:
|
|
self.litellm_check_batch_cost_last_run_timestamp.set(time.time())
|
|
self.litellm_check_batch_cost_jobs_polled.set(jobs_polled)
|
|
|
|
if processed_models:
|
|
for model, api_provider in processed_models:
|
|
self.litellm_check_batch_cost_jobs_processed_total.labels(
|
|
model=model or "",
|
|
api_provider=api_provider or "",
|
|
).inc()
|
|
except Exception as e:
|
|
verbose_logger.warning(f"Error recording check batch cost metrics: {e}")
|
|
|
|
def record_check_batch_cost_error(self, error_type: str):
|
|
try:
|
|
self.litellm_check_batch_cost_errors_total.labels(
|
|
error_type=error_type,
|
|
).inc()
|
|
except Exception as e:
|
|
verbose_logger.warning(
|
|
f"Error recording check batch cost error metric: {e}"
|
|
)
|
|
|
|
@staticmethod
|
|
def _get_exception_class_name(exception: Exception) -> str:
|
|
exception_class_name = ""
|
|
if hasattr(exception, "llm_provider"):
|
|
exception_class_name = getattr(exception, "llm_provider") or ""
|
|
|
|
# pretty print the provider name on prometheus
|
|
# eg. `openai` -> `Openai.`
|
|
if len(exception_class_name) >= 1:
|
|
exception_class_name = (
|
|
exception_class_name[0].upper() + exception_class_name[1:] + "."
|
|
)
|
|
|
|
exception_class_name += exception.__class__.__name__
|
|
return exception_class_name
|
|
|
|
async def log_success_fallback_event(
|
|
self, original_model_group: str, kwargs: dict, original_exception: Exception
|
|
):
|
|
"""
|
|
|
|
Logs a successful LLM fallback event on prometheus
|
|
|
|
"""
|
|
from litellm.litellm_core_utils.litellm_logging import (
|
|
StandardLoggingMetadata,
|
|
StandardLoggingPayloadSetup,
|
|
)
|
|
|
|
verbose_logger.debug(
|
|
"Prometheus: log_success_fallback_event, original_model_group: %s, kwargs: %s",
|
|
original_model_group,
|
|
kwargs,
|
|
)
|
|
_metadata_key = get_metadata_variable_name_from_kwargs(kwargs)
|
|
_metadata = kwargs.get(_metadata_key) or {}
|
|
standard_metadata: StandardLoggingMetadata = (
|
|
StandardLoggingPayloadSetup.get_standard_logging_metadata(
|
|
metadata=_metadata
|
|
)
|
|
)
|
|
_new_model = kwargs.get("model")
|
|
_tags = cast(List[str], kwargs.get("tags") or [])
|
|
|
|
enum_values = UserAPIKeyLabelValues(
|
|
requested_model=original_model_group,
|
|
fallback_model=_new_model,
|
|
hashed_api_key=standard_metadata["user_api_key_hash"],
|
|
api_key_alias=standard_metadata["user_api_key_alias"],
|
|
team=standard_metadata["user_api_key_team_id"],
|
|
team_alias=standard_metadata["user_api_key_team_alias"],
|
|
exception_status=str(getattr(original_exception, "status_code", None)),
|
|
exception_class=self._get_exception_class_name(original_exception),
|
|
tags=_tags,
|
|
)
|
|
PrometheusLogger._inc_labeled_counter(
|
|
self,
|
|
self.litellm_deployment_successful_fallbacks,
|
|
"litellm_deployment_successful_fallbacks",
|
|
enum_values,
|
|
label_context=PrometheusLabelFactoryContext(enum_values),
|
|
)
|
|
|
|
async def log_failure_fallback_event(
|
|
self, original_model_group: str, kwargs: dict, original_exception: Exception
|
|
):
|
|
"""
|
|
Logs a failed LLM fallback event on prometheus
|
|
"""
|
|
from litellm.litellm_core_utils.litellm_logging import (
|
|
StandardLoggingMetadata,
|
|
StandardLoggingPayloadSetup,
|
|
)
|
|
|
|
verbose_logger.debug(
|
|
"Prometheus: log_failure_fallback_event, original_model_group: %s, kwargs: %s",
|
|
original_model_group,
|
|
kwargs,
|
|
)
|
|
_new_model = kwargs.get("model")
|
|
_metadata_key = get_metadata_variable_name_from_kwargs(kwargs)
|
|
_metadata = kwargs.get(_metadata_key) or {}
|
|
_tags = cast(List[str], kwargs.get("tags") or [])
|
|
standard_metadata: StandardLoggingMetadata = (
|
|
StandardLoggingPayloadSetup.get_standard_logging_metadata(
|
|
metadata=_metadata
|
|
)
|
|
)
|
|
|
|
enum_values = UserAPIKeyLabelValues(
|
|
requested_model=original_model_group,
|
|
fallback_model=_new_model,
|
|
hashed_api_key=standard_metadata["user_api_key_hash"],
|
|
api_key_alias=standard_metadata["user_api_key_alias"],
|
|
team=standard_metadata["user_api_key_team_id"],
|
|
team_alias=standard_metadata["user_api_key_team_alias"],
|
|
exception_status=str(getattr(original_exception, "status_code", None)),
|
|
exception_class=self._get_exception_class_name(original_exception),
|
|
tags=_tags,
|
|
)
|
|
|
|
PrometheusLogger._inc_labeled_counter(
|
|
self,
|
|
self.litellm_deployment_failed_fallbacks,
|
|
"litellm_deployment_failed_fallbacks",
|
|
enum_values,
|
|
label_context=PrometheusLabelFactoryContext(enum_values),
|
|
)
|
|
|
|
def set_litellm_deployment_state(
|
|
self,
|
|
state: int,
|
|
litellm_model_name: str,
|
|
model_id: Optional[str],
|
|
api_base: Optional[str],
|
|
api_provider: str,
|
|
):
|
|
"""
|
|
Set the deployment state.
|
|
"""
|
|
### get labels
|
|
_labels = prometheus_label_factory(
|
|
supported_enum_labels=self.get_labels_for_metric(
|
|
metric_name="litellm_deployment_state"
|
|
),
|
|
enum_values=UserAPIKeyLabelValues(
|
|
litellm_model_name=litellm_model_name,
|
|
model_id=model_id,
|
|
api_base=api_base,
|
|
api_provider=api_provider,
|
|
),
|
|
)
|
|
self.litellm_deployment_state.labels(**_labels).set(state)
|
|
|
|
def set_deployment_healthy(
|
|
self,
|
|
litellm_model_name: str,
|
|
model_id: str,
|
|
api_base: str,
|
|
api_provider: str,
|
|
):
|
|
self.set_litellm_deployment_state(
|
|
0, litellm_model_name, model_id, api_base, api_provider
|
|
)
|
|
|
|
def set_deployment_partial_outage(
|
|
self,
|
|
litellm_model_name: str,
|
|
model_id: Optional[str],
|
|
api_base: Optional[str],
|
|
api_provider: str,
|
|
):
|
|
self.set_litellm_deployment_state(
|
|
1, litellm_model_name, model_id, api_base, api_provider
|
|
)
|
|
|
|
def set_deployment_complete_outage(
|
|
self,
|
|
litellm_model_name: str,
|
|
model_id: Optional[str],
|
|
api_base: Optional[str],
|
|
api_provider: str,
|
|
):
|
|
self.set_litellm_deployment_state(
|
|
2, litellm_model_name, model_id, api_base, api_provider
|
|
)
|
|
|
|
def increment_deployment_cooled_down(
|
|
self,
|
|
litellm_model_name: str,
|
|
model_id: str,
|
|
api_base: str,
|
|
api_provider: str,
|
|
exception_status: str,
|
|
):
|
|
"""
|
|
increment metric when litellm.Router / load balancing logic places a deployment in cool down
|
|
"""
|
|
self.litellm_deployment_cooled_down.labels(
|
|
_sanitize_prometheus_label_value(litellm_model_name),
|
|
_sanitize_prometheus_label_value(model_id),
|
|
_sanitize_prometheus_label_value(api_base),
|
|
_sanitize_prometheus_label_value(api_provider),
|
|
_sanitize_prometheus_label_value(exception_status),
|
|
).inc()
|
|
|
|
def increment_callback_logging_failure(
|
|
self,
|
|
callback_name: str,
|
|
):
|
|
"""
|
|
Increment metric when logging to a callback fails (e.g., s3_v2, langfuse, etc.)
|
|
"""
|
|
self.litellm_callback_logging_failures_metric.labels(
|
|
callback_name=callback_name
|
|
).inc()
|
|
|
|
def track_provider_remaining_budget(
|
|
self, provider: str, spend: float, budget_limit: float
|
|
):
|
|
"""
|
|
Track provider remaining budget in Prometheus
|
|
"""
|
|
self.litellm_provider_remaining_budget_metric.labels(provider).set(
|
|
self._safe_get_remaining_budget(
|
|
max_budget=budget_limit,
|
|
spend=spend,
|
|
)
|
|
)
|
|
|
|
def _safe_get_remaining_budget(
|
|
self, max_budget: Optional[float], spend: Optional[float]
|
|
) -> float:
|
|
if max_budget is None:
|
|
return float("inf")
|
|
|
|
if spend is None:
|
|
return max_budget
|
|
|
|
return max_budget - spend
|
|
|
|
async def _initialize_budget_metrics(
|
|
self,
|
|
data_fetch_function: Callable[..., Awaitable[Tuple[List[Any], Optional[int]]]],
|
|
set_metrics_function: Callable[[List[Any]], Awaitable[None]],
|
|
data_type: Literal["teams", "keys", "users", "orgs"],
|
|
):
|
|
"""
|
|
Generic method to initialize budget metrics for teams or API keys.
|
|
|
|
Args:
|
|
data_fetch_function: Function to fetch data with pagination.
|
|
set_metrics_function: Function to set metrics for the fetched data.
|
|
data_type: String representing the type of data ("teams" or "keys") for logging purposes.
|
|
"""
|
|
from litellm.proxy.proxy_server import prisma_client
|
|
|
|
if prisma_client is None:
|
|
return
|
|
|
|
try:
|
|
page = 1
|
|
page_size = 50
|
|
data, total_count = await data_fetch_function(
|
|
page_size=page_size, page=page
|
|
)
|
|
|
|
if total_count is None:
|
|
total_count = len(data)
|
|
|
|
# Calculate total pages needed
|
|
total_pages = (total_count + page_size - 1) // page_size
|
|
|
|
# Set metrics for first page of data
|
|
await set_metrics_function(data)
|
|
|
|
# Get and set metrics for remaining pages
|
|
for page in range(2, total_pages + 1):
|
|
data, _ = await data_fetch_function(page_size=page_size, page=page)
|
|
await set_metrics_function(data)
|
|
|
|
except Exception as e:
|
|
verbose_logger.exception(
|
|
f"Error initializing {data_type} budget metrics: {str(e)}"
|
|
)
|
|
|
|
async def _initialize_team_budget_metrics(self):
|
|
"""
|
|
Initialize team budget metrics by reusing the generic pagination logic.
|
|
"""
|
|
from litellm.proxy.management_endpoints.team_endpoints import (
|
|
get_paginated_teams,
|
|
)
|
|
from litellm.proxy.proxy_server import prisma_client
|
|
|
|
if prisma_client is None:
|
|
verbose_logger.debug(
|
|
"Prometheus: skipping team metrics initialization, DB not initialized"
|
|
)
|
|
return
|
|
|
|
async def fetch_teams(
|
|
page_size: int, page: int
|
|
) -> Tuple[List[LiteLLM_TeamTable], Optional[int]]:
|
|
teams, total_count = await get_paginated_teams(
|
|
prisma_client=prisma_client, page_size=page_size, page=page
|
|
)
|
|
if total_count is None:
|
|
total_count = len(teams)
|
|
return teams, total_count
|
|
|
|
await self._initialize_budget_metrics(
|
|
data_fetch_function=fetch_teams,
|
|
set_metrics_function=self._set_team_list_budget_metrics,
|
|
data_type="teams",
|
|
)
|
|
|
|
async def _initialize_api_key_budget_metrics(self):
|
|
"""
|
|
Initialize API key budget metrics by reusing the generic pagination logic.
|
|
"""
|
|
from litellm.constants import UI_SESSION_TOKEN_TEAM_ID
|
|
from litellm.proxy.management_endpoints.key_management_endpoints import (
|
|
_list_key_helper,
|
|
)
|
|
from litellm.proxy.proxy_server import prisma_client
|
|
|
|
if prisma_client is None:
|
|
verbose_logger.debug(
|
|
"Prometheus: skipping key metrics initialization, DB not initialized"
|
|
)
|
|
return
|
|
|
|
async def fetch_keys(page_size: int, page: int) -> Tuple[
|
|
List[Union[str, UserAPIKeyAuth, LiteLLM_DeletedVerificationToken]],
|
|
Optional[int],
|
|
]:
|
|
key_list_response = await _list_key_helper(
|
|
prisma_client=prisma_client,
|
|
page=page,
|
|
size=page_size,
|
|
user_id=None,
|
|
team_id=None,
|
|
key_alias=None,
|
|
key_hash=None,
|
|
exclude_team_id=UI_SESSION_TOKEN_TEAM_ID,
|
|
return_full_object=True,
|
|
organization_id=None,
|
|
)
|
|
keys = key_list_response.get("keys", [])
|
|
total_count = key_list_response.get("total_count")
|
|
if total_count is None:
|
|
total_count = len(keys)
|
|
return keys, total_count
|
|
|
|
await self._initialize_budget_metrics(
|
|
data_fetch_function=fetch_keys,
|
|
set_metrics_function=self._set_key_list_budget_metrics,
|
|
data_type="keys",
|
|
)
|
|
|
|
async def _initialize_user_budget_metrics(self):
|
|
"""
|
|
Initialize user budget metrics by reusing the generic pagination logic.
|
|
"""
|
|
from litellm.proxy.proxy_server import prisma_client
|
|
|
|
if prisma_client is None:
|
|
verbose_logger.debug(
|
|
"Prometheus: skipping user metrics initialization, DB not initialized"
|
|
)
|
|
return
|
|
|
|
async def fetch_users(
|
|
page_size: int, page: int
|
|
) -> Tuple[List[LiteLLM_UserTable], Optional[int]]:
|
|
skip = (page - 1) * page_size
|
|
users = await prisma_client.db.litellm_usertable.find_many(
|
|
skip=skip,
|
|
take=page_size,
|
|
order={"created_at": "desc"},
|
|
)
|
|
total_count = await prisma_client.db.litellm_usertable.count()
|
|
return users, total_count
|
|
|
|
await self._initialize_budget_metrics(
|
|
data_fetch_function=fetch_users,
|
|
set_metrics_function=self._set_user_list_budget_metrics,
|
|
data_type="users",
|
|
)
|
|
|
|
async def _initialize_org_budget_metrics(self):
|
|
"""
|
|
Initialize org budget metrics by reusing the generic pagination logic.
|
|
"""
|
|
from litellm.proxy.proxy_server import prisma_client
|
|
|
|
if prisma_client is None:
|
|
verbose_logger.debug(
|
|
"Prometheus: skipping org metrics initialization, DB not initialized"
|
|
)
|
|
return
|
|
|
|
async def fetch_orgs(page_size: int, page: int) -> Tuple[list, Optional[int]]:
|
|
skip = (page - 1) * page_size
|
|
orgs = await prisma_client.db.litellm_organizationtable.find_many(
|
|
skip=skip,
|
|
take=page_size,
|
|
order={"created_at": "desc"},
|
|
include={"litellm_budget_table": True},
|
|
)
|
|
total_count = await prisma_client.db.litellm_organizationtable.count()
|
|
return orgs, total_count
|
|
|
|
await self._initialize_budget_metrics(
|
|
data_fetch_function=fetch_orgs,
|
|
set_metrics_function=self._set_org_list_budget_metrics,
|
|
data_type="orgs",
|
|
)
|
|
|
|
async def initialize_remaining_budget_metrics(self):
|
|
"""
|
|
Handler for initializing remaining budget metrics for all teams to avoid metric discrepancies.
|
|
|
|
Runs when prometheus logger starts up.
|
|
|
|
- If redis cache is available, we use the pod lock manager to acquire a lock and initialize the metrics.
|
|
- Ensures only one pod emits the metrics at a time.
|
|
- If redis cache is not available, we initialize the metrics directly.
|
|
"""
|
|
from litellm.constants import PROMETHEUS_EMIT_BUDGET_METRICS_JOB_NAME
|
|
from litellm.proxy.proxy_server import proxy_logging_obj
|
|
|
|
pod_lock_manager = proxy_logging_obj.db_spend_update_writer.pod_lock_manager
|
|
|
|
# if using redis, ensure only one pod emits the metrics at a time
|
|
if pod_lock_manager and pod_lock_manager.redis_cache:
|
|
if await pod_lock_manager.acquire_lock(
|
|
cronjob_id=PROMETHEUS_EMIT_BUDGET_METRICS_JOB_NAME
|
|
):
|
|
try:
|
|
await self._initialize_remaining_budget_metrics()
|
|
finally:
|
|
await pod_lock_manager.release_lock(
|
|
cronjob_id=PROMETHEUS_EMIT_BUDGET_METRICS_JOB_NAME
|
|
)
|
|
else:
|
|
# if not using redis, initialize the metrics directly
|
|
await self._initialize_remaining_budget_metrics()
|
|
|
|
async def _initialize_remaining_budget_metrics(self):
|
|
"""
|
|
Helper to initialize remaining budget metrics for all teams, API keys, and users.
|
|
"""
|
|
verbose_logger.debug("Emitting key, team, user, org budget metrics....")
|
|
await self._initialize_team_budget_metrics()
|
|
await self._initialize_api_key_budget_metrics()
|
|
await self._initialize_user_budget_metrics()
|
|
await self._initialize_org_budget_metrics()
|
|
await self._initialize_user_and_team_count_metrics()
|
|
|
|
async def _initialize_user_and_team_count_metrics(self):
|
|
"""
|
|
Initialize user and team count metrics by querying the database.
|
|
|
|
Updates:
|
|
- litellm_total_users: Total count of users in the database
|
|
- litellm_teams_count: Total count of teams in the database
|
|
"""
|
|
from litellm.proxy.proxy_server import prisma_client
|
|
|
|
if prisma_client is None:
|
|
verbose_logger.debug(
|
|
"Prometheus: skipping user/team count metrics initialization, DB not initialized"
|
|
)
|
|
return
|
|
|
|
try:
|
|
# Get total user count
|
|
total_users = await prisma_client.db.litellm_usertable.count()
|
|
self.litellm_total_users_metric.set(total_users)
|
|
verbose_logger.debug(
|
|
f"Prometheus: set litellm_total_users to {total_users}"
|
|
)
|
|
|
|
# Get total team count
|
|
total_teams = await prisma_client.db.litellm_teamtable.count()
|
|
self.litellm_teams_count_metric.set(total_teams)
|
|
verbose_logger.debug(
|
|
f"Prometheus: set litellm_teams_count to {total_teams}"
|
|
)
|
|
except Exception as e:
|
|
verbose_logger.exception(
|
|
f"Error initializing user/team count metrics: {str(e)}"
|
|
)
|
|
|
|
async def _set_key_list_budget_metrics(
|
|
self, keys: List[Union[str, UserAPIKeyAuth]]
|
|
):
|
|
"""Helper function to set budget metrics for a list of keys"""
|
|
for key in keys:
|
|
if isinstance(key, UserAPIKeyAuth):
|
|
self._set_key_budget_metrics(key)
|
|
|
|
async def _set_team_list_budget_metrics(self, teams: List[LiteLLM_TeamTable]):
|
|
"""Helper function to set budget metrics for a list of teams"""
|
|
for team in teams:
|
|
self._set_team_budget_metrics(team)
|
|
|
|
async def _set_user_list_budget_metrics(self, users: List[LiteLLM_UserTable]):
|
|
"""Helper function to set budget metrics for a list of users"""
|
|
for user in users:
|
|
self._set_user_budget_metrics(user)
|
|
|
|
async def _set_org_list_budget_metrics(self, orgs: list):
|
|
"""Helper function to set budget metrics for a list of orgs"""
|
|
for org in orgs:
|
|
budget_table = getattr(org, "litellm_budget_table", None)
|
|
self._set_org_budget_metrics(
|
|
org_id=org.organization_id or "",
|
|
org_alias=org.organization_alias or "",
|
|
spend=org.spend or 0.0,
|
|
max_budget=budget_table.max_budget if budget_table else None,
|
|
budget_reset_at=(
|
|
getattr(budget_table, "budget_reset_at", None)
|
|
if budget_table
|
|
else None
|
|
),
|
|
)
|
|
|
|
async def _set_team_budget_metrics_after_api_request(
|
|
self,
|
|
user_api_team: Optional[str],
|
|
user_api_team_alias: Optional[str],
|
|
team_spend: Optional[float],
|
|
team_max_budget: Optional[float],
|
|
response_cost: float,
|
|
):
|
|
"""
|
|
Set team budget metrics after an LLM API request
|
|
|
|
- Assemble a LiteLLM_TeamTable object
|
|
- looks up team info from db if not available in metadata
|
|
- Set team budget metrics
|
|
"""
|
|
if user_api_team:
|
|
team_object = await self._assemble_team_object(
|
|
team_id=user_api_team,
|
|
team_alias=user_api_team_alias or "",
|
|
spend=team_spend,
|
|
max_budget=team_max_budget,
|
|
response_cost=response_cost,
|
|
)
|
|
|
|
self._set_team_budget_metrics(team_object)
|
|
|
|
async def _assemble_team_object(
|
|
self,
|
|
team_id: str,
|
|
team_alias: str,
|
|
spend: Optional[float],
|
|
max_budget: Optional[float],
|
|
response_cost: float,
|
|
) -> LiteLLM_TeamTable:
|
|
"""
|
|
Assemble a LiteLLM_TeamTable object
|
|
|
|
for fields not available in metadata, we fetch from db
|
|
Fields not available in metadata:
|
|
- `budget_reset_at`
|
|
"""
|
|
from litellm.proxy.auth.auth_checks import get_team_object
|
|
from litellm.proxy.proxy_server import prisma_client, user_api_key_cache
|
|
|
|
_total_team_spend = (spend or 0) + response_cost
|
|
team_object = LiteLLM_TeamTable(
|
|
team_id=team_id,
|
|
team_alias=team_alias,
|
|
spend=_total_team_spend,
|
|
max_budget=max_budget,
|
|
)
|
|
try:
|
|
team_info = await get_team_object(
|
|
team_id=team_id,
|
|
prisma_client=prisma_client,
|
|
user_api_key_cache=user_api_key_cache,
|
|
)
|
|
except Exception as e:
|
|
verbose_logger.debug(
|
|
f"[Non-Blocking] Prometheus: Error getting team info: {str(e)}"
|
|
)
|
|
return team_object
|
|
|
|
if team_info:
|
|
team_object.budget_reset_at = team_info.budget_reset_at
|
|
if team_object.max_budget is None and team_info.max_budget is not None:
|
|
team_object.max_budget = team_info.max_budget
|
|
|
|
return team_object
|
|
|
|
def _set_team_budget_metrics(
|
|
self,
|
|
team: LiteLLM_TeamTable,
|
|
):
|
|
"""
|
|
Set team budget metrics for a single team
|
|
|
|
- Remaining Budget
|
|
- Max Budget
|
|
- Budget Reset At
|
|
"""
|
|
enum_values = UserAPIKeyLabelValues(
|
|
team=team.team_id,
|
|
team_alias=team.team_alias or "",
|
|
)
|
|
|
|
_labels = prometheus_label_factory(
|
|
supported_enum_labels=self.get_labels_for_metric(
|
|
metric_name="litellm_remaining_team_budget_metric"
|
|
),
|
|
enum_values=enum_values,
|
|
)
|
|
self.litellm_remaining_team_budget_metric.labels(**_labels).set(
|
|
self._safe_get_remaining_budget(
|
|
max_budget=team.max_budget,
|
|
spend=team.spend,
|
|
)
|
|
)
|
|
|
|
if team.max_budget is not None:
|
|
_labels = prometheus_label_factory(
|
|
supported_enum_labels=self.get_labels_for_metric(
|
|
metric_name="litellm_team_max_budget_metric"
|
|
),
|
|
enum_values=enum_values,
|
|
)
|
|
self.litellm_team_max_budget_metric.labels(**_labels).set(team.max_budget)
|
|
|
|
if team.budget_reset_at is not None:
|
|
_labels = prometheus_label_factory(
|
|
supported_enum_labels=self.get_labels_for_metric(
|
|
metric_name="litellm_team_budget_remaining_hours_metric"
|
|
),
|
|
enum_values=enum_values,
|
|
)
|
|
self.litellm_team_budget_remaining_hours_metric.labels(**_labels).set(
|
|
self._get_remaining_hours_for_budget_reset(
|
|
budget_reset_at=team.budget_reset_at
|
|
)
|
|
)
|
|
|
|
async def _set_org_budget_metrics_after_api_request(
|
|
self,
|
|
org_id: Optional[str],
|
|
response_cost: float,
|
|
):
|
|
"""
|
|
Set org budget metrics after an LLM API request
|
|
|
|
- Fetches org info via cache (get_org_object)
|
|
- Sets org budget metrics
|
|
"""
|
|
if not org_id:
|
|
return
|
|
|
|
from litellm.proxy.auth.auth_checks import get_org_object
|
|
from litellm.proxy.proxy_server import prisma_client, user_api_key_cache
|
|
|
|
if prisma_client is None:
|
|
return
|
|
|
|
try:
|
|
org_info = await get_org_object(
|
|
org_id=org_id,
|
|
prisma_client=prisma_client,
|
|
user_api_key_cache=user_api_key_cache,
|
|
include_budget_table=True,
|
|
)
|
|
except Exception as e:
|
|
verbose_logger.debug(
|
|
f"[Non-Blocking] Prometheus: Error getting org info: {str(e)}"
|
|
)
|
|
return
|
|
|
|
if org_info is None:
|
|
return
|
|
|
|
org_alias = org_info.organization_alias or ""
|
|
_total_org_spend = (org_info.spend or 0.0) + response_cost
|
|
budget_table = org_info.litellm_budget_table
|
|
max_budget = budget_table.max_budget if budget_table else None
|
|
budget_reset_at = (
|
|
getattr(budget_table, "budget_reset_at", None) if budget_table else None
|
|
)
|
|
|
|
self._set_org_budget_metrics(
|
|
org_id=org_id,
|
|
org_alias=org_alias,
|
|
spend=_total_org_spend,
|
|
max_budget=max_budget,
|
|
budget_reset_at=budget_reset_at,
|
|
)
|
|
|
|
def _set_org_budget_metrics(
|
|
self,
|
|
org_id: str,
|
|
org_alias: str,
|
|
spend: float,
|
|
max_budget: Optional[float],
|
|
budget_reset_at: Optional[datetime],
|
|
):
|
|
"""
|
|
Set org budget metrics for a single org
|
|
|
|
- Remaining Budget
|
|
- Max Budget
|
|
- Budget Reset At
|
|
"""
|
|
enum_values = UserAPIKeyLabelValues(
|
|
org_id=org_id,
|
|
org_alias=org_alias,
|
|
)
|
|
|
|
_labels = prometheus_label_factory(
|
|
supported_enum_labels=self.get_labels_for_metric(
|
|
metric_name="litellm_remaining_org_budget_metric"
|
|
),
|
|
enum_values=enum_values,
|
|
)
|
|
self.litellm_remaining_org_budget_metric.labels(**_labels).set(
|
|
self._safe_get_remaining_budget(
|
|
max_budget=max_budget,
|
|
spend=spend,
|
|
)
|
|
)
|
|
|
|
if max_budget is not None:
|
|
_labels = prometheus_label_factory(
|
|
supported_enum_labels=self.get_labels_for_metric(
|
|
metric_name="litellm_org_max_budget_metric"
|
|
),
|
|
enum_values=enum_values,
|
|
)
|
|
self.litellm_org_max_budget_metric.labels(**_labels).set(max_budget)
|
|
|
|
if budget_reset_at is not None:
|
|
_labels = prometheus_label_factory(
|
|
supported_enum_labels=self.get_labels_for_metric(
|
|
metric_name="litellm_org_budget_remaining_hours_metric"
|
|
),
|
|
enum_values=enum_values,
|
|
)
|
|
self.litellm_org_budget_remaining_hours_metric.labels(**_labels).set(
|
|
self._get_remaining_hours_for_budget_reset(
|
|
budget_reset_at=budget_reset_at
|
|
)
|
|
)
|
|
|
|
def _set_key_budget_metrics(self, user_api_key_dict: UserAPIKeyAuth):
|
|
"""
|
|
Set virtual key budget metrics
|
|
|
|
- Remaining Budget
|
|
- Max Budget
|
|
- Budget Reset At
|
|
"""
|
|
enum_values = UserAPIKeyLabelValues(
|
|
hashed_api_key=user_api_key_dict.token,
|
|
api_key_alias=user_api_key_dict.key_alias or "",
|
|
)
|
|
_labels = prometheus_label_factory(
|
|
supported_enum_labels=self.get_labels_for_metric(
|
|
metric_name="litellm_remaining_api_key_budget_metric"
|
|
),
|
|
enum_values=enum_values,
|
|
)
|
|
self.litellm_remaining_api_key_budget_metric.labels(**_labels).set(
|
|
self._safe_get_remaining_budget(
|
|
max_budget=user_api_key_dict.max_budget,
|
|
spend=user_api_key_dict.spend,
|
|
)
|
|
)
|
|
|
|
if user_api_key_dict.max_budget is not None:
|
|
_labels = prometheus_label_factory(
|
|
supported_enum_labels=self.get_labels_for_metric(
|
|
metric_name="litellm_api_key_max_budget_metric"
|
|
),
|
|
enum_values=enum_values,
|
|
)
|
|
self.litellm_api_key_max_budget_metric.labels(**_labels).set(
|
|
user_api_key_dict.max_budget
|
|
)
|
|
|
|
if user_api_key_dict.budget_reset_at is not None:
|
|
self.litellm_api_key_budget_remaining_hours_metric.labels(**_labels).set(
|
|
self._get_remaining_hours_for_budget_reset(
|
|
budget_reset_at=user_api_key_dict.budget_reset_at
|
|
)
|
|
)
|
|
|
|
async def _set_api_key_budget_metrics_after_api_request(
|
|
self,
|
|
user_api_key: Optional[str],
|
|
user_api_key_alias: Optional[str],
|
|
response_cost: float,
|
|
key_max_budget: Optional[float],
|
|
key_spend: Optional[float],
|
|
):
|
|
if user_api_key:
|
|
user_api_key_dict = await self._assemble_key_object(
|
|
user_api_key=user_api_key,
|
|
user_api_key_alias=user_api_key_alias or "",
|
|
key_max_budget=key_max_budget,
|
|
key_spend=key_spend,
|
|
response_cost=response_cost,
|
|
)
|
|
self._set_key_budget_metrics(user_api_key_dict)
|
|
|
|
async def _assemble_key_object(
|
|
self,
|
|
user_api_key: str,
|
|
user_api_key_alias: str,
|
|
key_max_budget: Optional[float],
|
|
key_spend: Optional[float],
|
|
response_cost: float,
|
|
) -> UserAPIKeyAuth:
|
|
"""
|
|
Assemble a UserAPIKeyAuth object
|
|
"""
|
|
from litellm.proxy.auth.auth_checks import get_key_object
|
|
from litellm.proxy.proxy_server import prisma_client, user_api_key_cache
|
|
|
|
_total_key_spend = (key_spend or 0) + response_cost
|
|
user_api_key_dict = UserAPIKeyAuth(
|
|
token=user_api_key,
|
|
key_alias=user_api_key_alias,
|
|
max_budget=key_max_budget,
|
|
spend=_total_key_spend,
|
|
)
|
|
try:
|
|
if user_api_key_dict.token:
|
|
key_object = await get_key_object(
|
|
hashed_token=user_api_key_dict.token,
|
|
prisma_client=prisma_client,
|
|
user_api_key_cache=user_api_key_cache,
|
|
)
|
|
if key_object:
|
|
user_api_key_dict.budget_reset_at = key_object.budget_reset_at
|
|
except Exception as e:
|
|
verbose_logger.debug(
|
|
f"[Non-Blocking] Prometheus: Error getting key info: {str(e)}"
|
|
)
|
|
|
|
return user_api_key_dict
|
|
|
|
async def _set_user_budget_metrics_after_api_request(
|
|
self,
|
|
user_id: Optional[str],
|
|
user_spend: Optional[float],
|
|
user_max_budget: Optional[float],
|
|
response_cost: float,
|
|
):
|
|
"""
|
|
Set user budget metrics after an LLM API request
|
|
|
|
- Assemble a LiteLLM_UserTable object
|
|
- looks up user info from db if not available in metadata
|
|
- Set user budget metrics
|
|
"""
|
|
if user_id:
|
|
user_object = await self._assemble_user_object(
|
|
user_id=user_id,
|
|
spend=user_spend,
|
|
max_budget=user_max_budget,
|
|
response_cost=response_cost,
|
|
)
|
|
|
|
self._set_user_budget_metrics(user_object)
|
|
|
|
async def _assemble_user_object(
|
|
self,
|
|
user_id: str,
|
|
spend: Optional[float],
|
|
max_budget: Optional[float],
|
|
response_cost: float,
|
|
) -> LiteLLM_UserTable:
|
|
"""
|
|
Assemble a LiteLLM_UserTable object
|
|
|
|
for fields not available in metadata, we fetch from db
|
|
Fields not available in metadata:
|
|
- `budget_reset_at`
|
|
"""
|
|
from litellm.proxy.auth.auth_checks import get_user_object
|
|
from litellm.proxy.proxy_server import prisma_client, user_api_key_cache
|
|
|
|
_total_user_spend = (spend or 0) + response_cost
|
|
user_object = LiteLLM_UserTable(
|
|
user_id=user_id,
|
|
spend=_total_user_spend,
|
|
max_budget=max_budget,
|
|
)
|
|
try:
|
|
# Note: Setting check_db_only=True bypasses cache and hits DB on every request,
|
|
# causing huge latency increase and CPU spikes. Keep check_db_only=False.
|
|
user_info = await get_user_object(
|
|
user_id=user_id,
|
|
prisma_client=prisma_client,
|
|
user_api_key_cache=user_api_key_cache,
|
|
user_id_upsert=False,
|
|
check_db_only=False,
|
|
)
|
|
except Exception as e:
|
|
verbose_logger.debug(
|
|
f"[Non-Blocking] Prometheus: Error getting user info: {str(e)}"
|
|
)
|
|
return user_object
|
|
|
|
if user_info:
|
|
user_object.budget_reset_at = user_info.budget_reset_at
|
|
if user_object.max_budget is None and user_info.max_budget is not None:
|
|
user_object.max_budget = user_info.max_budget
|
|
|
|
return user_object
|
|
|
|
def _set_user_budget_metrics(
|
|
self,
|
|
user: LiteLLM_UserTable,
|
|
):
|
|
"""
|
|
Set user budget metrics for a single user
|
|
|
|
- Remaining Budget
|
|
- Max Budget
|
|
- Budget Reset At
|
|
"""
|
|
enum_values = UserAPIKeyLabelValues(
|
|
user=user.user_id,
|
|
)
|
|
|
|
_labels = prometheus_label_factory(
|
|
supported_enum_labels=self.get_labels_for_metric(
|
|
metric_name="litellm_remaining_user_budget_metric"
|
|
),
|
|
enum_values=enum_values,
|
|
)
|
|
self.litellm_remaining_user_budget_metric.labels(**_labels).set(
|
|
self._safe_get_remaining_budget(
|
|
max_budget=user.max_budget,
|
|
spend=user.spend,
|
|
)
|
|
)
|
|
|
|
if user.max_budget is not None:
|
|
_labels = prometheus_label_factory(
|
|
supported_enum_labels=self.get_labels_for_metric(
|
|
metric_name="litellm_user_max_budget_metric"
|
|
),
|
|
enum_values=enum_values,
|
|
)
|
|
self.litellm_user_max_budget_metric.labels(**_labels).set(user.max_budget)
|
|
|
|
if user.budget_reset_at is not None:
|
|
_labels = prometheus_label_factory(
|
|
supported_enum_labels=self.get_labels_for_metric(
|
|
metric_name="litellm_user_budget_remaining_hours_metric"
|
|
),
|
|
enum_values=enum_values,
|
|
)
|
|
self.litellm_user_budget_remaining_hours_metric.labels(**_labels).set(
|
|
self._get_remaining_hours_for_budget_reset(
|
|
budget_reset_at=user.budget_reset_at
|
|
)
|
|
)
|
|
|
|
def _get_remaining_hours_for_budget_reset(self, budget_reset_at: datetime) -> float:
|
|
"""
|
|
Get remaining hours for budget reset
|
|
"""
|
|
return (
|
|
budget_reset_at - datetime.now(budget_reset_at.tzinfo)
|
|
).total_seconds() / 3600
|
|
|
|
def _safe_duration_seconds(
|
|
self,
|
|
start_time: Any,
|
|
end_time: Any,
|
|
) -> Optional[float]:
|
|
"""
|
|
Compute the duration in seconds between two objects.
|
|
|
|
Returns the duration as a float if both start and end are instances of datetime,
|
|
otherwise returns None.
|
|
"""
|
|
if isinstance(start_time, datetime) and isinstance(end_time, datetime):
|
|
return (end_time - start_time).total_seconds()
|
|
return None
|
|
|
|
@staticmethod
|
|
def initialize_budget_metrics_cron_job(scheduler: AsyncIOScheduler):
|
|
"""
|
|
Initialize budget metrics as a cron job. This job runs every `PROMETHEUS_BUDGET_METRICS_REFRESH_INTERVAL_MINUTES` minutes.
|
|
|
|
It emits the current remaining budget metrics for all Keys and Teams.
|
|
"""
|
|
from litellm.constants import PROMETHEUS_BUDGET_METRICS_REFRESH_INTERVAL_MINUTES
|
|
|
|
prometheus_loggers: List[CustomLogger] = (
|
|
litellm.logging_callback_manager.get_custom_loggers_for_type(
|
|
callback_type=PrometheusLogger
|
|
)
|
|
)
|
|
# we need to get the initialized prometheus logger instance(s) and call logger.initialize_remaining_budget_metrics() on them
|
|
verbose_logger.debug("found %s prometheus loggers", len(prometheus_loggers))
|
|
if len(prometheus_loggers) > 0:
|
|
prometheus_logger = cast(PrometheusLogger, prometheus_loggers[0])
|
|
verbose_logger.debug(
|
|
"Initializing remaining budget metrics as a cron job executing every %s minutes"
|
|
% PROMETHEUS_BUDGET_METRICS_REFRESH_INTERVAL_MINUTES
|
|
)
|
|
scheduler.add_job(
|
|
prometheus_logger.initialize_remaining_budget_metrics,
|
|
"interval",
|
|
minutes=PROMETHEUS_BUDGET_METRICS_REFRESH_INTERVAL_MINUTES,
|
|
# REMOVED jitter parameter - major cause of memory leak
|
|
id="prometheus_budget_metrics_job",
|
|
replace_existing=True,
|
|
)
|
|
|
|
@staticmethod
|
|
def _mount_metrics_endpoint():
|
|
"""
|
|
Mount the Prometheus metrics endpoint with optional authentication.
|
|
|
|
Args:
|
|
require_auth (bool, optional): Whether to require authentication for the metrics endpoint.
|
|
Defaults to False.
|
|
"""
|
|
from prometheus_client import make_asgi_app
|
|
|
|
from litellm._logging import verbose_proxy_logger
|
|
from litellm.proxy.proxy_server import app
|
|
|
|
# Create metrics ASGI app
|
|
if "PROMETHEUS_MULTIPROC_DIR" in os.environ:
|
|
from prometheus_client import CollectorRegistry, multiprocess
|
|
|
|
registry = CollectorRegistry()
|
|
multiprocess.MultiProcessCollector(registry)
|
|
metrics_app = make_asgi_app(registry)
|
|
else:
|
|
metrics_app = make_asgi_app()
|
|
|
|
# Mount the metrics app to the app
|
|
app.mount("/metrics", metrics_app)
|
|
verbose_proxy_logger.debug(
|
|
"Starting Prometheus Metrics on /metrics (no authentication)"
|
|
)
|
|
|
|
|
|
def _prometheus_labels_from_context(
|
|
supported_enum_labels: List[str],
|
|
ctx: PrometheusLabelFactoryContext,
|
|
) -> Dict[str, Optional[str]]:
|
|
filtered_labels: Dict[str, Optional[str]] = {
|
|
label: ctx._sanitized_enum[label]
|
|
for label in supported_enum_labels
|
|
if label in ctx._sanitized_enum
|
|
}
|
|
|
|
if UserAPIKeyLabelNames.END_USER.value in filtered_labels:
|
|
filtered_labels[UserAPIKeyLabelNames.END_USER.value] = (
|
|
ctx.get_resolved_end_user()
|
|
)
|
|
|
|
for sk, val in ctx._custom_by_sanitized_key.items():
|
|
if sk in supported_enum_labels:
|
|
filtered_labels[sk] = val
|
|
|
|
for k, v in ctx._tag_labels.items():
|
|
if k in supported_enum_labels:
|
|
filtered_labels[k] = v
|
|
|
|
for label in supported_enum_labels:
|
|
if label not in filtered_labels:
|
|
filtered_labels[label] = None
|
|
|
|
return filtered_labels
|
|
|
|
|
|
def prometheus_label_factory(
|
|
supported_enum_labels: List[str],
|
|
enum_values: UserAPIKeyLabelValues,
|
|
tag: Optional[str] = None,
|
|
*,
|
|
label_context: Optional[PrometheusLabelFactoryContext] = None,
|
|
) -> dict:
|
|
"""
|
|
Returns a dictionary of label + values for prometheus.
|
|
|
|
Ensures end_user param is not sent to prometheus if it is not supported.
|
|
|
|
When ``label_context`` is provided, it must have been built from the same
|
|
``enum_values`` object; work is amortized (single model_dump, tag map, etc.).
|
|
"""
|
|
if label_context is not None:
|
|
if label_context.enum_values is not enum_values:
|
|
raise ValueError(
|
|
"label_context.enum_values must be the same object as enum_values"
|
|
)
|
|
return _prometheus_labels_from_context(supported_enum_labels, label_context)
|
|
|
|
# Extract dictionary from Pydantic object
|
|
enum_dict = enum_values.model_dump()
|
|
|
|
# Filter supported labels and sanitize values to prevent breaking
|
|
# the Prometheus text format (e.g. U+2028 Line Separator in label values)
|
|
filtered_labels = {
|
|
label: _sanitize_prometheus_label_value(value)
|
|
for label, value in enum_dict.items()
|
|
if label in supported_enum_labels
|
|
}
|
|
|
|
if UserAPIKeyLabelNames.END_USER.value in filtered_labels:
|
|
get_end_user_id_for_cost_tracking = _get_cached_end_user_id_for_cost_tracking()
|
|
|
|
filtered_labels["end_user"] = get_end_user_id_for_cost_tracking(
|
|
litellm_params={"user_api_key_end_user_id": enum_values.end_user},
|
|
service_type="prometheus",
|
|
)
|
|
|
|
if enum_values.custom_metadata_labels is not None:
|
|
for key, value in enum_values.custom_metadata_labels.items():
|
|
# check sanitized key
|
|
sanitized_key = _sanitize_prometheus_label_name(key)
|
|
if sanitized_key in supported_enum_labels:
|
|
filtered_labels[sanitized_key] = _sanitize_prometheus_label_value(value)
|
|
|
|
# Add custom tags if configured
|
|
if enum_values.tags is not None:
|
|
custom_tag_labels = get_custom_labels_from_tags(enum_values.tags)
|
|
for key, value in custom_tag_labels.items():
|
|
if key in supported_enum_labels:
|
|
filtered_labels[key] = _sanitize_prometheus_label_value(value)
|
|
|
|
for label in supported_enum_labels:
|
|
if label not in filtered_labels:
|
|
filtered_labels[label] = None
|
|
|
|
return filtered_labels
|
|
|
|
|
|
def get_custom_labels_from_metadata(metadata: dict) -> Dict[str, str]:
|
|
"""
|
|
Get custom labels from metadata
|
|
"""
|
|
keys = litellm.custom_prometheus_metadata_labels
|
|
if keys is None or len(keys) == 0:
|
|
return {}
|
|
|
|
result: Dict[str, str] = {}
|
|
|
|
for key in keys:
|
|
# Split the dot notation key into parts
|
|
original_key = key
|
|
key = key.replace("metadata.", "", 1) if key.startswith("metadata.") else key
|
|
|
|
keys_parts = key.split(".")
|
|
# Traverse through the dictionary using the parts
|
|
value: Any = metadata
|
|
for part in keys_parts:
|
|
if isinstance(value, dict):
|
|
value = value.get(part, None) # Get the value, return None if not found
|
|
else:
|
|
value = None
|
|
if value is None:
|
|
break
|
|
|
|
if value is not None and isinstance(value, str):
|
|
result[original_key.replace(".", "_")] = value
|
|
|
|
return result
|
|
|
|
|
|
def _tag_matches_wildcard_configured_pattern(
|
|
tags: Sequence[str], configured_tag: str
|
|
) -> bool:
|
|
"""
|
|
Check if any of the request tags matches a wildcard configured pattern
|
|
|
|
Args:
|
|
tags: List[str] - The request tags
|
|
configured_tag: str - The configured tag
|
|
|
|
Returns:
|
|
bool - True if any of the request tags matches the configured tag, False otherwise
|
|
|
|
e.g.
|
|
tags = ["User-Agent: curl/7.68.0", "User-Agent: python-requests/2.28.1", "prod"]
|
|
configured_tag = "User-Agent: curl/*"
|
|
_tag_matches_wildcard_configured_pattern(tags=tags, configured_tag=configured_tag) # True
|
|
|
|
configured_tag = "User-Agent: python-requests/*"
|
|
_tag_matches_wildcard_configured_pattern(tags=tags, configured_tag=configured_tag) # True
|
|
|
|
configured_tag = "gm"
|
|
_tag_matches_wildcard_configured_pattern(tags=tags, configured_tag=configured_tag) # False
|
|
"""
|
|
import re
|
|
|
|
from litellm.router_utils.pattern_match_deployments import PatternMatchRouter
|
|
|
|
pattern_router = PatternMatchRouter()
|
|
regex_pattern = pattern_router._pattern_to_regex(configured_tag)
|
|
return any(re.match(pattern=regex_pattern, string=tag) for tag in tags)
|
|
|
|
|
|
def get_custom_labels_from_tags(tags: Sequence[str]) -> Dict[str, str]:
|
|
"""
|
|
Get custom labels from tags based on admin configuration.
|
|
|
|
Supports both exact matches and wildcard patterns:
|
|
- Exact match: "prod" matches "prod" exactly
|
|
- Wildcard pattern: "User-Agent: curl/*" matches "User-Agent: curl/7.68.0"
|
|
|
|
Reuses PatternMatchRouter for wildcard pattern matching.
|
|
|
|
Returns dict of label_name: "true" if the tag matches the configured tag, "false" otherwise
|
|
|
|
{
|
|
"tag_User-Agent_curl": "true",
|
|
"tag_User-Agent_python_requests": "false",
|
|
"tag_Environment_prod": "true",
|
|
"tag_Environment_dev": "false",
|
|
"tag_Service_api_gateway_v2": "true",
|
|
"tag_Service_web_app_v1": "false",
|
|
}
|
|
"""
|
|
|
|
from litellm.types.integrations.prometheus import _sanitize_prometheus_label_name
|
|
|
|
configured_tags = litellm.custom_prometheus_tags
|
|
if configured_tags is None or len(configured_tags) == 0:
|
|
return {}
|
|
|
|
result: Dict[str, str] = {}
|
|
|
|
for configured_tag in configured_tags:
|
|
label_name = _sanitize_prometheus_label_name(f"tag_{configured_tag}")
|
|
|
|
# Check for exact match first (backwards compatibility)
|
|
if configured_tag in tags:
|
|
result[label_name] = "true"
|
|
continue
|
|
|
|
# Use PatternMatchRouter for wildcard pattern matching
|
|
if "*" in configured_tag and _tag_matches_wildcard_configured_pattern(
|
|
tags=tags, configured_tag=configured_tag
|
|
):
|
|
result[label_name] = "true"
|
|
continue
|
|
|
|
# No match found
|
|
result[label_name] = "false"
|
|
|
|
return result
|