mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
feat(prometheus): expose per-key and per-team rate limit allowed and used gauges (#39236)
Co-authored-by: yassin <yassin@berri.ai> Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
81277252e1
commit
6d0367ce35
5 changed files with 395 additions and 2 deletions
|
|
@ -8,6 +8,7 @@ import math
|
|||
import os
|
||||
import sys
|
||||
from collections.abc import Awaitable, Callable, Mapping, Sequence
|
||||
from dataclasses import replace
|
||||
from datetime import datetime, timedelta
|
||||
from typing import TYPE_CHECKING, Any, Final, Literal, Protocol, TypeVar, cast
|
||||
|
||||
|
|
@ -58,6 +59,7 @@ from litellm.types.utils import (
|
|||
|
||||
if TYPE_CHECKING:
|
||||
from apscheduler.schedulers.asyncio import AsyncIOScheduler
|
||||
from prometheus_client import Gauge
|
||||
from prometheus_client.metrics import MetricWrapperBase
|
||||
|
||||
from litellm.router import Router
|
||||
|
|
@ -476,6 +478,30 @@ class PrometheusLogger(CustomLogger):
|
|||
labelnames=self.get_labels_for_metric("litellm_remaining_api_key_tokens_for_model"),
|
||||
)
|
||||
|
||||
self.litellm_api_key_rate_limit_allowed_metric = self._gauge_factory(
|
||||
"litellm_api_key_rate_limit_allowed_metric",
|
||||
"Configured rate limit for the API Key in the current window (rpm_limit / tpm_limit), by rate_limit_type",
|
||||
labelnames=self.get_labels_for_metric("litellm_api_key_rate_limit_allowed_metric"),
|
||||
)
|
||||
|
||||
self.litellm_api_key_rate_limit_used_metric = self._gauge_factory(
|
||||
"litellm_api_key_rate_limit_used_metric",
|
||||
"Requests or tokens the API Key has consumed in the current rate limit window, by rate_limit_type",
|
||||
labelnames=self.get_labels_for_metric("litellm_api_key_rate_limit_used_metric"),
|
||||
)
|
||||
|
||||
self.litellm_team_rate_limit_allowed_metric = self._gauge_factory(
|
||||
"litellm_team_rate_limit_allowed_metric",
|
||||
"Configured rate limit for the Team in the current window (team rpm_limit / tpm_limit), by rate_limit_type",
|
||||
labelnames=self.get_labels_for_metric("litellm_team_rate_limit_allowed_metric"),
|
||||
)
|
||||
|
||||
self.litellm_team_rate_limit_used_metric = self._gauge_factory(
|
||||
"litellm_team_rate_limit_used_metric",
|
||||
"Requests or tokens the Team has consumed in the current rate limit window, by rate_limit_type",
|
||||
labelnames=self.get_labels_for_metric("litellm_team_rate_limit_used_metric"),
|
||||
)
|
||||
|
||||
########################################
|
||||
# LLM API Deployment Metrics / analytics
|
||||
########################################
|
||||
|
|
@ -1475,6 +1501,11 @@ class PrometheusLogger(CustomLogger):
|
|||
model_id=enum_values.model_id,
|
||||
)
|
||||
|
||||
self._set_key_and_team_rate_limit_metrics(
|
||||
standard_logging_payload=standard_logging_payload, # pyright: ignore[reportArgumentType] # isinstance(dict) above narrows the TypedDict to dict[Unknown, Unknown]
|
||||
enum_values=enum_values,
|
||||
)
|
||||
|
||||
# set latency metrics
|
||||
self._set_latency_metrics(
|
||||
kwargs=kwargs,
|
||||
|
|
@ -2002,17 +2033,102 @@ class PrometheusLogger(CustomLogger):
|
|||
"""
|
||||
if standard_logging_payload is None:
|
||||
return None
|
||||
return PrometheusLogger._get_int_from_v3_rate_limit_headers(
|
||||
standard_logging_payload=standard_logging_payload,
|
||||
header_name=f"x-ratelimit-model_per_key-remaining-{rate_limit_type}",
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _get_int_from_v3_rate_limit_headers(
|
||||
standard_logging_payload: StandardLoggingPayload,
|
||||
header_name: str,
|
||||
) -> int | None:
|
||||
hidden_params: Final = standard_logging_payload.get("hidden_params")
|
||||
if hidden_params is None:
|
||||
return None
|
||||
additional_headers: Final = hidden_params.get("additional_headers")
|
||||
additional_headers: Final[Mapping[str, object] | None] = hidden_params.get("additional_headers")
|
||||
if additional_headers is None:
|
||||
return None
|
||||
value: Final = dict(additional_headers).get(f"x-ratelimit-model_per_key-remaining-{rate_limit_type}")
|
||||
value: Final = additional_headers.get(header_name)
|
||||
if isinstance(value, bool) or not isinstance(value, int):
|
||||
return None
|
||||
return value
|
||||
|
||||
def _set_key_and_team_rate_limit_metrics(
|
||||
self,
|
||||
standard_logging_payload: StandardLoggingPayload,
|
||||
enum_values: UserAPIKeyLabelValues,
|
||||
) -> None:
|
||||
"""
|
||||
Export the key-level and team-level RPM / TPM limit and current window
|
||||
usage from the ``x-ratelimit-{api_key,team}-{limit,remaining}-*``
|
||||
headers the v3 rate limiter mirrors into the logging payload. The
|
||||
limiter already read these counters (from Redis when configured) on
|
||||
the request path, so no extra store lookup happens here. Descriptors
|
||||
without a configured limit emit no header, so their series is removed
|
||||
rather than left at the value from before the limit was dropped.
|
||||
"""
|
||||
descriptor_gauges: Final[
|
||||
tuple[tuple[Literal["api_key", "team"], DEFINED_PROMETHEUS_METRICS, Gauge, Gauge], ...]
|
||||
] = (
|
||||
(
|
||||
"api_key",
|
||||
"litellm_api_key_rate_limit_allowed_metric",
|
||||
self.litellm_api_key_rate_limit_allowed_metric,
|
||||
self.litellm_api_key_rate_limit_used_metric,
|
||||
),
|
||||
(
|
||||
"team",
|
||||
"litellm_team_rate_limit_allowed_metric",
|
||||
self.litellm_team_rate_limit_allowed_metric,
|
||||
self.litellm_team_rate_limit_used_metric,
|
||||
),
|
||||
)
|
||||
for descriptor_key, metric_name, allowed_gauge, used_gauge in descriptor_gauges:
|
||||
for rate_limit_type in ("requests", "tokens"):
|
||||
self._set_rate_limit_allowed_and_used_gauges(
|
||||
standard_logging_payload=standard_logging_payload,
|
||||
enum_values=enum_values,
|
||||
descriptor_key=descriptor_key,
|
||||
metric_name=metric_name,
|
||||
allowed_gauge=allowed_gauge,
|
||||
used_gauge=used_gauge,
|
||||
rate_limit_type=rate_limit_type,
|
||||
)
|
||||
|
||||
def _set_rate_limit_allowed_and_used_gauges(
|
||||
self,
|
||||
standard_logging_payload: StandardLoggingPayload,
|
||||
enum_values: UserAPIKeyLabelValues,
|
||||
descriptor_key: Literal["api_key", "team"],
|
||||
metric_name: DEFINED_PROMETHEUS_METRICS,
|
||||
allowed_gauge: Gauge,
|
||||
used_gauge: Gauge,
|
||||
rate_limit_type: Literal["requests", "tokens"],
|
||||
) -> None:
|
||||
limit: Final = self._get_int_from_v3_rate_limit_headers(
|
||||
standard_logging_payload=standard_logging_payload,
|
||||
header_name=f"x-ratelimit-{descriptor_key}-limit-{rate_limit_type}",
|
||||
)
|
||||
remaining: Final = self._get_int_from_v3_rate_limit_headers(
|
||||
standard_logging_payload=standard_logging_payload,
|
||||
header_name=f"x-ratelimit-{descriptor_key}-remaining-{rate_limit_type}",
|
||||
)
|
||||
labelled_values: Final = replace(enum_values, rate_limit_type=rate_limit_type)
|
||||
labelnames: Final = self.get_labels_for_metric(metric_name)
|
||||
labels: Final = prometheus_label_factory(
|
||||
supported_enum_labels=labelnames,
|
||||
enum_values=labelled_values,
|
||||
label_context=PrometheusLabelFactoryContext(labelled_values),
|
||||
)
|
||||
if limit is None or remaining is None:
|
||||
label_values: Final = tuple(labels.get(label) for label in labelnames)
|
||||
self._bounded_prometheus_series_tracker.remove_series(allowed_gauge, label_values)
|
||||
self._bounded_prometheus_series_tracker.remove_series(used_gauge, label_values)
|
||||
return
|
||||
allowed_gauge.labels(**labels).set(limit)
|
||||
used_gauge.labels(**labels).set(limit - remaining)
|
||||
|
||||
def _set_virtual_key_rate_limit_metrics(
|
||||
self,
|
||||
user_api_key: str | None,
|
||||
|
|
|
|||
|
|
@ -60,6 +60,10 @@ class BoundedPrometheusSeriesTracker:
|
|||
break
|
||||
del series[tracked_label_values]
|
||||
|
||||
def remove_series(self, metric: object, label_values: tuple[str | None, ...]) -> bool:
|
||||
"""Drop one child series, True when it is gone (removed or never existed)."""
|
||||
return self._remove_metric_child(metric, label_values)
|
||||
|
||||
def _should_run_ttl_cleanup(
|
||||
self,
|
||||
metric_name: str,
|
||||
|
|
|
|||
|
|
@ -270,6 +270,10 @@ DEFINED_PROMETHEUS_METRICS = Literal[
|
|||
"litellm_deployment_rpm_limit",
|
||||
"litellm_remaining_api_key_requests_for_model",
|
||||
"litellm_remaining_api_key_tokens_for_model",
|
||||
"litellm_api_key_rate_limit_allowed_metric",
|
||||
"litellm_api_key_rate_limit_used_metric",
|
||||
"litellm_team_rate_limit_allowed_metric",
|
||||
"litellm_team_rate_limit_used_metric",
|
||||
"litellm_llm_api_failed_requests_metric",
|
||||
"litellm_callback_logging_failures_metric",
|
||||
"litellm_in_flight_requests",
|
||||
|
|
@ -775,6 +779,22 @@ class PrometheusMetricLabels:
|
|||
UserAPIKeyLabelNames.MODEL_ID.value,
|
||||
]
|
||||
|
||||
litellm_api_key_rate_limit_allowed_metric: ClassVar[tuple[str, ...]] = (
|
||||
UserAPIKeyLabelNames.API_KEY_HASH.value,
|
||||
UserAPIKeyLabelNames.API_KEY_ALIAS.value,
|
||||
UserAPIKeyLabelNames.RATE_LIMIT_TYPE.value,
|
||||
)
|
||||
|
||||
litellm_api_key_rate_limit_used_metric = litellm_api_key_rate_limit_allowed_metric
|
||||
|
||||
litellm_team_rate_limit_allowed_metric: ClassVar[tuple[str, ...]] = (
|
||||
UserAPIKeyLabelNames.TEAM.value,
|
||||
UserAPIKeyLabelNames.TEAM_ALIAS.value,
|
||||
UserAPIKeyLabelNames.RATE_LIMIT_TYPE.value,
|
||||
)
|
||||
|
||||
litellm_team_rate_limit_used_metric = litellm_team_rate_limit_allowed_metric
|
||||
|
||||
litellm_llm_api_failed_requests_metric = [
|
||||
UserAPIKeyLabelNames.END_USER.value,
|
||||
UserAPIKeyLabelNames.API_KEY_HASH.value,
|
||||
|
|
|
|||
|
|
@ -93,6 +93,7 @@ async def test_async_post_call_success_hook_includes_client_ip_user_agent():
|
|||
logger._increment_token_metrics = MagicMock()
|
||||
logger._increment_remaining_budget_metrics = AsyncMock()
|
||||
logger._set_virtual_key_rate_limit_metrics = MagicMock()
|
||||
logger._set_key_and_team_rate_limit_metrics = MagicMock()
|
||||
logger._set_latency_metrics = MagicMock()
|
||||
logger.set_llm_deployment_success_metrics = MagicMock()
|
||||
logger._increment_cache_metrics = MagicMock()
|
||||
|
|
|
|||
|
|
@ -13,6 +13,7 @@ Covers two follow-up gaps to the unified rate-limit error work:
|
|||
429s don't silently break when the new class lands.
|
||||
"""
|
||||
|
||||
from collections.abc import Mapping
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
|
@ -471,3 +472,254 @@ def test_should_ignore_non_int_v3_header_values(bad_value):
|
|||
logger.litellm_remaining_api_key_tokens_for_model.labels.return_value.set.assert_called_once_with(
|
||||
sys.maxsize
|
||||
)
|
||||
|
||||
|
||||
KEY_AND_TEAM_RATE_LIMIT_METRICS = (
|
||||
"litellm_api_key_rate_limit_allowed_metric",
|
||||
"litellm_api_key_rate_limit_used_metric",
|
||||
"litellm_team_rate_limit_allowed_metric",
|
||||
"litellm_team_rate_limit_used_metric",
|
||||
)
|
||||
|
||||
|
||||
def _clear_prometheus_registry() -> None:
|
||||
from prometheus_client import REGISTRY
|
||||
|
||||
for collector in list(REGISTRY._collector_to_names.keys()):
|
||||
try:
|
||||
REGISTRY.unregister(collector)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
def _collected_samples(metric_name: str) -> dict[tuple[tuple[str, str], ...], float]:
|
||||
from prometheus_client import REGISTRY
|
||||
|
||||
return {
|
||||
tuple(sorted(sample.labels.items())): sample.value
|
||||
for metric in REGISTRY.collect()
|
||||
for sample in metric.samples
|
||||
if sample.name == metric_name
|
||||
}
|
||||
|
||||
|
||||
def _success_kwargs_with_rate_limit_headers(additional_headers: Mapping[str, object] | None) -> dict[str, object]:
|
||||
return {
|
||||
"model": "claude-haiku-4-5",
|
||||
"litellm_params": {"metadata": {}},
|
||||
"standard_logging_object": {
|
||||
"id": "t",
|
||||
"call_type": "completion",
|
||||
"response_cost": 0.001,
|
||||
"status": "success",
|
||||
"total_tokens": 20,
|
||||
"prompt_tokens": 15,
|
||||
"completion_tokens": 5,
|
||||
"startTime": 1.0,
|
||||
"endTime": 2.0,
|
||||
"completionStartTime": 1.5,
|
||||
"model": "claude-haiku-4-5",
|
||||
"model_id": "model-123",
|
||||
"model_group": "anthropic-haiku-4-5",
|
||||
"api_base": "https://api.anthropic.com",
|
||||
"custom_llm_provider": "anthropic",
|
||||
"request_tags": [],
|
||||
"end_user": None,
|
||||
"cache_hit": False,
|
||||
"stream": False,
|
||||
"response": None,
|
||||
"model_parameters": None,
|
||||
"metadata": {
|
||||
"user_api_key_hash": "key-hash",
|
||||
"user_api_key_alias": "key-alias",
|
||||
"user_api_key_team_id": "team-id",
|
||||
"user_api_key_team_alias": "team-alias",
|
||||
"user_api_key_user_id": "u",
|
||||
"user_api_key_user_email": "e@x.com",
|
||||
"user_api_key_org_id": None,
|
||||
"user_api_key_org_alias": None,
|
||||
"requester_metadata": None,
|
||||
"user_api_key_end_user_id": None,
|
||||
"usage_object": None,
|
||||
},
|
||||
"hidden_params": {
|
||||
"litellm_overhead_time_ms": None,
|
||||
"additional_headers": additional_headers,
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
async def _run_success_event(
|
||||
additional_headers: Mapping[str, object] | None, logger: PrometheusLogger | None = None
|
||||
) -> None:
|
||||
import datetime
|
||||
|
||||
now = datetime.datetime.now()
|
||||
await (logger or PrometheusLogger()).async_log_success_event(
|
||||
_success_kwargs_with_rate_limit_headers(additional_headers), None, now, now
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_should_emit_key_and_team_rate_limit_allowed_and_used_from_v3_headers():
|
||||
"""
|
||||
LIT-1672: the v3 limiter mirrors ``x-ratelimit-{api_key,team}-{limit,remaining}-*``
|
||||
into the logging payload. The gauges must expose the configured limit as-is
|
||||
and the window consumption as ``limit - remaining`` for each key / team
|
||||
dimension, split by ``rate_limit_type``.
|
||||
"""
|
||||
_clear_prometheus_registry()
|
||||
try:
|
||||
await _run_success_event(
|
||||
{
|
||||
"x-ratelimit-api_key-limit-requests": 10,
|
||||
"x-ratelimit-api_key-remaining-requests": 7,
|
||||
"x-ratelimit-api_key-limit-tokens": 20000,
|
||||
"x-ratelimit-api_key-remaining-tokens": 19947,
|
||||
"x-ratelimit-team-limit-requests": 50,
|
||||
"x-ratelimit-team-remaining-requests": 47,
|
||||
"x-ratelimit-team-limit-tokens": 40000,
|
||||
"x-ratelimit-team-remaining-tokens": 39960,
|
||||
"x-ratelimit-model_per_key-limit-requests": 5,
|
||||
"x-ratelimit-model_per_key-remaining-requests": 1,
|
||||
}
|
||||
)
|
||||
|
||||
key_requests = (
|
||||
("api_key_alias", "key-alias"),
|
||||
("hashed_api_key", "key-hash"),
|
||||
("rate_limit_type", "requests"),
|
||||
)
|
||||
key_tokens = (
|
||||
("api_key_alias", "key-alias"),
|
||||
("hashed_api_key", "key-hash"),
|
||||
("rate_limit_type", "tokens"),
|
||||
)
|
||||
team_requests = (
|
||||
("rate_limit_type", "requests"),
|
||||
("team", "team-id"),
|
||||
("team_alias", "team-alias"),
|
||||
)
|
||||
team_tokens = (
|
||||
("rate_limit_type", "tokens"),
|
||||
("team", "team-id"),
|
||||
("team_alias", "team-alias"),
|
||||
)
|
||||
|
||||
assert _collected_samples("litellm_api_key_rate_limit_allowed_metric") == {
|
||||
key_requests: 10,
|
||||
key_tokens: 20000,
|
||||
}
|
||||
assert _collected_samples("litellm_api_key_rate_limit_used_metric") == {
|
||||
key_requests: 3,
|
||||
key_tokens: 53,
|
||||
}
|
||||
assert _collected_samples("litellm_team_rate_limit_allowed_metric") == {
|
||||
team_requests: 50,
|
||||
team_tokens: 40000,
|
||||
}
|
||||
assert _collected_samples("litellm_team_rate_limit_used_metric") == {
|
||||
team_requests: 3,
|
||||
team_tokens: 40,
|
||||
}
|
||||
finally:
|
||||
_clear_prometheus_registry()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_should_emit_only_the_dimensions_the_limiter_enforced():
|
||||
"""
|
||||
A key with only ``rpm_limit`` set and no team limits produces only the
|
||||
key/requests headers, so no tokens series and no team series may appear
|
||||
(a phantom 0 or sys.maxsize series would misreport an unlimited dimension).
|
||||
"""
|
||||
_clear_prometheus_registry()
|
||||
try:
|
||||
await _run_success_event(
|
||||
{
|
||||
"x-ratelimit-api_key-limit-requests": 10,
|
||||
"x-ratelimit-api_key-remaining-requests": 10,
|
||||
}
|
||||
)
|
||||
|
||||
key_requests = (
|
||||
("api_key_alias", "key-alias"),
|
||||
("hashed_api_key", "key-hash"),
|
||||
("rate_limit_type", "requests"),
|
||||
)
|
||||
assert _collected_samples("litellm_api_key_rate_limit_allowed_metric") == {key_requests: 10}
|
||||
assert _collected_samples("litellm_api_key_rate_limit_used_metric") == {key_requests: 0}
|
||||
assert _collected_samples("litellm_team_rate_limit_allowed_metric") == {}
|
||||
assert _collected_samples("litellm_team_rate_limit_used_metric") == {}
|
||||
finally:
|
||||
_clear_prometheus_registry()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_should_drop_key_and_team_series_once_the_limiter_stops_reporting_a_limit():
|
||||
"""
|
||||
Removing a key's ``rpm_limit`` / ``tpm_limit`` (or a team's ``tpm_limit``)
|
||||
makes the v3 limiter stop emitting that descriptor's headers on later
|
||||
requests. The old allowed/used samples must disappear instead of keeping
|
||||
a limit that no longer exists on the scrape.
|
||||
"""
|
||||
_clear_prometheus_registry()
|
||||
try:
|
||||
logger = PrometheusLogger()
|
||||
await _run_success_event(
|
||||
{
|
||||
"x-ratelimit-api_key-limit-requests": 10,
|
||||
"x-ratelimit-api_key-remaining-requests": 7,
|
||||
"x-ratelimit-api_key-limit-tokens": 20000,
|
||||
"x-ratelimit-api_key-remaining-tokens": 19947,
|
||||
"x-ratelimit-team-limit-requests": 50,
|
||||
"x-ratelimit-team-remaining-requests": 47,
|
||||
"x-ratelimit-team-limit-tokens": 40000,
|
||||
"x-ratelimit-team-remaining-tokens": 39960,
|
||||
},
|
||||
logger=logger,
|
||||
)
|
||||
await _run_success_event(
|
||||
{
|
||||
"x-ratelimit-team-limit-requests": 50,
|
||||
"x-ratelimit-team-remaining-requests": 46,
|
||||
},
|
||||
logger=logger,
|
||||
)
|
||||
|
||||
team_requests = (
|
||||
("rate_limit_type", "requests"),
|
||||
("team", "team-id"),
|
||||
("team_alias", "team-alias"),
|
||||
)
|
||||
assert _collected_samples("litellm_api_key_rate_limit_allowed_metric") == {}
|
||||
assert _collected_samples("litellm_api_key_rate_limit_used_metric") == {}
|
||||
assert _collected_samples("litellm_team_rate_limit_allowed_metric") == {team_requests: 50}
|
||||
assert _collected_samples("litellm_team_rate_limit_used_metric") == {team_requests: 4}
|
||||
finally:
|
||||
_clear_prometheus_registry()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize(
|
||||
"additional_headers",
|
||||
[
|
||||
None,
|
||||
{"x-ratelimit-model_per_key-remaining-requests": 42},
|
||||
{"x-ratelimit-api_key-limit-requests": 10},
|
||||
{"x-ratelimit-api_key-limit-requests": "10", "x-ratelimit-api_key-remaining-requests": "7"},
|
||||
{"x-ratelimit-team-limit-tokens": True, "x-ratelimit-team-remaining-tokens": 5},
|
||||
],
|
||||
)
|
||||
async def test_should_emit_no_key_or_team_rate_limit_series_without_a_complete_int_pair(
|
||||
additional_headers,
|
||||
):
|
||||
_clear_prometheus_registry()
|
||||
try:
|
||||
await _run_success_event(additional_headers)
|
||||
|
||||
for metric_name in KEY_AND_TEAM_RATE_LIMIT_METRICS:
|
||||
assert _collected_samples(metric_name) == {}, metric_name
|
||||
finally:
|
||||
_clear_prometheus_registry()
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue