From 58e13cbb6352b8b1efd2c8de5a8914dbe16f00b9 Mon Sep 17 00:00:00 2001 From: DanBrima Date: Mon, 28 Sep 2026 15:21:51 +0300 Subject: [PATCH] feat(prometheus): expose team-scoped rate limit gauges Adds team-and-model-scoped gauges for configured and remaining RPM/TPM, sourced from the v3 rate limiter's response headers and updated from successful request logging payloads. - registers the gauges with configurable labels - retires series when a limit is removed or a team alias is renamed, skipping retirement under multiprocess collection where worker-backed samples cannot be withdrawn - tracks the last team labelset instead of scanning the registry Rebased onto main: the prometheus tests moved from tests/test_litellm/integrations/ to tests/unit/integrations/, so the new suite lands at the current path. Squashed from the original 10 commits; the source changes applied to main without conflict. Co-Authored-By: Claude Opus 5 (1M context) --- litellm/integrations/prometheus.py | 277 ++++++++- litellm/types/integrations/prometheus.py | 18 + .../test_prometheus_client_ip_user_agent.py | 1 + ...test_prometheus_team_rate_limit_metrics.py | 549 ++++++++++++++++++ 4 files changed, 831 insertions(+), 14 deletions(-) create mode 100644 tests/unit/integrations/test_prometheus_team_rate_limit_metrics.py diff --git a/litellm/integrations/prometheus.py b/litellm/integrations/prometheus.py index c7bf291a887..f20921340c9 100644 --- a/litellm/integrations/prometheus.py +++ b/litellm/integrations/prometheus.py @@ -7,7 +7,7 @@ import asyncio import math import os import sys -from collections.abc import Awaitable, Callable, Mapping, Sequence +from collections.abc import Awaitable, Callable, Mapping, MutableMapping, Sequence from dataclasses import replace from datetime import datetime, timedelta from types import MappingProxyType @@ -184,6 +184,13 @@ class _ExcludedLabelMetric: ) return self._metric.labels(*kept_values) if kept_values else self._metric + def remove(self, *labelvalues: str) -> None: + kept_values: Final = tuple( + value for name, value in zip(self._original_labelnames, labelvalues) if name not in self._excluded_labels + ) + if kept_values: + self._metric.remove(*kept_values) + _MetricLike: TypeAlias = "NoOpMetric | _ExcludedLabelMetric | MetricWrapperBase" @@ -206,6 +213,53 @@ def _get_budget_metrics_per_request_timeout() -> float: return parsed +class _LabeledGauge(Protocol): + """ + Structural type shared by ``prometheus_client.Gauge`` and the no-op / + label-excluding wrappers above. + + ``remove`` is declared even though retirement is performed by + ``BoundedPrometheusSeriesTracker``: the tracker takes an untyped metric and + reports an ``AttributeError`` as "not removed" rather than raising, so a + wrapper that stopped exposing ``remove`` would silently retire nothing. + Requiring it here is what keeps that a type error instead. + """ + + def labels(self, *labelvalues: str) -> _LabeledGauge: ... + + def set(self, value: float) -> None: ... + + def remove(self, *labelvalues: str) -> None: ... + + +_TEAM_RATE_LIMIT_GAUGE_SPECS: Final[ + tuple[ + tuple[ + DEFINED_PROMETHEUS_METRICS, + Literal["remaining", "limit"], + Literal["requests", "tokens"], + ], + ..., + ] +] = ( + ("litellm_remaining_team_requests_for_model", "remaining", "requests"), + ("litellm_remaining_team_tokens_for_model", "remaining", "tokens"), + ("litellm_team_rpm_limit", "limit", "requests"), + ("litellm_team_tpm_limit", "limit", "tokens"), +) + + +def _series_retirement_supported() -> bool: + """ + ``prometheus_client`` refuses to remove a labelset in multiprocess mode and + warns when asked, because each worker owns its own mmap file and cannot + retire a series another worker wrote. Retirement is therefore a + single-process capability, and attempting it under multiprocess collection + would only emit warnings while leaving the sample in place. + """ + return not ("PROMETHEUS_MULTIPROC_DIR" in os.environ or "prometheus_multiproc_dir" in os.environ) + + def _get_proxy_llm_router() -> Router | None: try: from litellm.proxy.proxy_server import llm_router @@ -299,6 +353,11 @@ class PrometheusLogger(CustomLogger): _custom_buckets: Final = litellm.prometheus_latency_buckets self.latency_buckets = tuple(_custom_buckets) if _custom_buckets is not None else LATENCY_BUCKETS self._bounded_prometheus_series_tracker = BoundedPrometheusSeriesTracker() + # Last labelset emitted per (metric, team, model), so a renamed team's + # previous series can be retired without scanning the registry. + self._team_series_label_values: MutableMapping[ # mutable-ok: per-process emission state, rewritten as teams are renamed + tuple[str, str, str], tuple[str, ...] + ] = {} # Create metric factory functions self._counter_factory = self._create_metric_factory(Counter) @@ -572,6 +631,34 @@ class PrometheusLogger(CustomLogger): labelnames=self.get_labels_for_metric("litellm_team_rate_limit_used_metric"), ) + ######################################## + # LiteLLM Team rate limit metrics + ######################################## + + self.litellm_remaining_team_requests_for_model = self._gauge_factory( + "litellm_remaining_team_requests_for_model", + "Remaining Requests team can make for model (model based rpm limit on team)", + labelnames=self.get_labels_for_metric("litellm_remaining_team_requests_for_model"), + ) + + self.litellm_remaining_team_tokens_for_model = self._gauge_factory( + "litellm_remaining_team_tokens_for_model", + "Remaining Tokens team can make for model (model based tpm limit on team)", + labelnames=self.get_labels_for_metric("litellm_remaining_team_tokens_for_model"), + ) + + self.litellm_team_rpm_limit = self._gauge_factory( + "litellm_team_rpm_limit", + "Configured RPM limit for team + model (model based rpm limit on team)", + labelnames=self.get_labels_for_metric("litellm_team_rpm_limit"), + ) + + self.litellm_team_tpm_limit = self._gauge_factory( + "litellm_team_tpm_limit", + "Configured TPM limit for team + model (model based tpm limit on team)", + labelnames=self.get_labels_for_metric("litellm_team_tpm_limit"), + ) + ######################################## # LLM API Deployment Metrics / analytics ######################################## @@ -1480,6 +1567,14 @@ class PrometheusLogger(CustomLogger): model_id=enum_values.model_id, ) + # set team rpm/tpm metrics for the requested model + self._set_team_rate_limit_metrics( + user_api_team=user_api_team, + user_api_team_alias=user_api_team_alias, + model_group=standard_logging_payload["model_group"], + standard_logging_payload=standard_logging_payload, + ) + self._set_key_and_team_rate_limit_metrics( standard_logging_payload=standard_logging_payload, # pyright: ignore[reportArgumentType] # isinstance(dict) above narrows the TypedDict to dict[Unknown, Unknown] enum_values=enum_values, @@ -2047,18 +2142,21 @@ class PrometheusLogger(CustomLogger): series.set(math.nan if capture_rate is None else capture_rate) @staticmethod - def _get_remaining_from_v3_rate_limit_headers( + def _get_v3_rate_limit_header( standard_logging_payload: StandardLoggingPayload | None, + descriptor_key: Literal["model_per_key", "model_per_team"], + value_type: Literal["remaining", "limit"], rate_limit_type: Literal["requests", "tokens"], ) -> int | None: """ - Read the per-(key, model) remaining value emitted by the v3 rate - limiter (``parallel_request_limiter_v3.py``), which writes - ``x-ratelimit-model_per_key-remaining-{requests,tokens}`` into - ``standard_logging_object.hidden_params.additional_headers`` instead - of the ``litellm-key-remaining-*`` metadata keys the legacy limiter - sets. The header carries no model group; it always refers to this - request's model group, which is what the gauges are labeled with. + Read a per-(scope, model) value emitted by the v3 rate limiter + (``parallel_request_limiter_v3.py``), which writes + ``x-ratelimit-{descriptor_key}-{remaining,limit}-{requests,tokens}`` + into ``standard_logging_object.hidden_params.additional_headers`` + instead of the ``litellm-key-remaining-*`` metadata keys the legacy + limiter sets. The header carries no model group; it always refers to + this request's model group, which is what the gauges are labeled + with. A scope with no configured limit produces no header at all. Values are written in-process as plain ints (never HTTP-serialized strings), so anything else is rejected rather than coerced. """ @@ -2066,7 +2164,7 @@ class PrometheusLogger(CustomLogger): return None return PrometheusLogger._get_int_from_v3_rate_limit_headers( standard_logging_payload=standard_logging_payload, - header_name=f"x-ratelimit-model_per_key-remaining-{rate_limit_type}", + header_name=f"x-ratelimit-{descriptor_key}-{value_type}-{rate_limit_type}", ) @staticmethod @@ -2181,15 +2279,21 @@ class PrometheusLogger(CustomLogger): remaining_requests = metadata.get(remaining_requests_variable_name) if remaining_requests is None: - remaining_requests = self._get_remaining_from_v3_rate_limit_headers( - standard_logging_payload=standard_logging_payload, rate_limit_type="requests" + remaining_requests = self._get_v3_rate_limit_header( + standard_logging_payload=standard_logging_payload, + descriptor_key="model_per_key", + value_type="remaining", + rate_limit_type="requests", ) if remaining_requests is None: remaining_requests = sys.maxsize remaining_tokens = metadata.get(remaining_tokens_variable_name) if remaining_tokens is None: - remaining_tokens = self._get_remaining_from_v3_rate_limit_headers( - standard_logging_payload=standard_logging_payload, rate_limit_type="tokens" + remaining_tokens = self._get_v3_rate_limit_header( + standard_logging_payload=standard_logging_payload, + descriptor_key="model_per_key", + value_type="remaining", + rate_limit_type="tokens", ) if remaining_tokens is None: remaining_tokens = sys.maxsize @@ -2220,6 +2324,151 @@ class PrometheusLogger(CustomLogger): ) self.litellm_remaining_api_key_tokens_for_model.labels(**tokens_labels).set(remaining_tokens) + def _set_team_rate_limit_metrics( + self, + user_api_team: str | None, + user_api_team_alias: str | None, + model_group: str | None, + standard_logging_payload: StandardLoggingPayload | None, + ) -> None: + """ + Emit the per-(team, model) rate limit gauges from the values the v3 + rate limiter already computed for its ``model_per_team`` descriptor + and shipped to the client as ``x-ratelimit-model_per_team-*`` + headers. A team with no per-model limit configured for the requested + model produces no header, and therefore no series, which matches how + the per-key gauges behave. + """ + if user_api_team is None: + return + + enum_values: Final = UserAPIKeyLabelValues( + team=user_api_team, + team_alias=user_api_team_alias, + model=model_group, + custom_metadata_labels=get_custom_labels_from_metadata( + metadata=_get_combined_custom_metadata_from_standard_logging_payload( + standard_logging_payload=standard_logging_payload + ) + ), + ) + label_context: Final = PrometheusLabelFactoryContext(enum_values) + + for metric_name, value_type, rate_limit_type in _TEAM_RATE_LIMIT_GAUGE_SPECS: + self._sync_team_rate_limit_gauge( + gauge=getattr(self, metric_name), + metric_name=metric_name, + value=self._get_v3_rate_limit_header( + standard_logging_payload=standard_logging_payload, + descriptor_key="model_per_team", + value_type=value_type, + rate_limit_type=rate_limit_type, + ), + enum_values=enum_values, + label_context=label_context, + ) + + def _sync_team_rate_limit_gauge( + self, + gauge: _LabeledGauge, + metric_name: DEFINED_PROMETHEUS_METRICS, + value: int | None, + enum_values: UserAPIKeyLabelValues, + label_context: PrometheusLabelFactoryContext, + ) -> None: + """ + Set the gauge, or drop its child series when this team has no limit + configured for this model. Prometheus keeps a child series for the + life of the process once emitted, so without the drop a team whose + limit is removed would keep publishing the last remaining/limit + values it ever saw, and alerts would evaluate against a number no + longer being enforced. + + Superseded aliases are swept on both paths, because a team can be + renamed and then have its limit removed before it sends another + limited request, which would otherwise strand the pre-rename series. + + Both drops go through ``BoundedPrometheusSeriesTracker.remove_series``, + the same call the key/team rate limit gauges use, rather than a second + removal path of this file's own. + """ + labelnames: Final = self.get_labels_for_metric(metric_name) + if UserAPIKeyLabelNames.TEAM.value not in labelnames: + # Without a team label the gauge collapses to one sample shared by + # every team, which cannot attribute a limit to anyone and cannot + # be retired. Publishing nothing beats publishing a number that + # silently belongs to whichever team wrote it last. + return + + labels: Final = prometheus_label_factory( + supported_enum_labels=labelnames, + enum_values=enum_values, + label_context=label_context, + ) + label_values: Final = tuple(labels.get(name, "") for name in labelnames) + can_retire: Final = _series_retirement_supported() + if can_retire: + self._drop_superseded_team_series( + gauge=gauge, metric_name=metric_name, labels=labels, label_values=label_values + ) + if value is not None: + gauge.labels(*label_values).set(value) + return + + if not can_retire: + return + + self._forget_team_series(metric_name=metric_name, labels=labels) + # Removal goes through the shared tracker so this file has one path + # that drops a child series. A labelset with no child is the common + # case here -- the team never had a limit for this model -- and the + # tracker already treats that as removed rather than an error. + self._bounded_prometheus_series_tracker.remove_series(gauge, label_values) + + def _drop_superseded_team_series( + self, + gauge: _LabeledGauge, + metric_name: DEFINED_PROMETHEUS_METRICS, + labels: Mapping[str, str], + label_values: tuple[str, ...], + ) -> None: + """ + Retire the series this team and model last published under a different + alias. Renaming a team changes ``team_alias``, which starts a new + series, and the old one would otherwise keep publishing the values it + held at rename time, double counting the team on any sum over ``team``. + + The previously emitted labelset is remembered per (metric, team, model) + rather than found by scanning the registry. A scan would cost every + team request work proportional to the total number of team series ever + emitted, which any authenticated caller could amplify by sending + ordinary traffic. + """ + identity: Final = ( + metric_name, + labels.get(UserAPIKeyLabelNames.TEAM.value, ""), + labels.get(UserAPIKeyLabelNames.v1_LITELLM_MODEL_NAME.value, ""), + ) + previous: Final = self._team_series_label_values.get(identity) + if previous is not None and previous != label_values: + self._bounded_prometheus_series_tracker.remove_series(gauge, previous) + self._team_series_label_values[identity] = label_values + + def _forget_team_series( + self, + metric_name: DEFINED_PROMETHEUS_METRICS, + labels: Mapping[str, str], + ) -> None: + """Stop tracking a (metric, team, model) whose series has been dropped.""" + self._team_series_label_values.pop( + ( + metric_name, + labels.get(UserAPIKeyLabelNames.TEAM.value, ""), + labels.get(UserAPIKeyLabelNames.v1_LITELLM_MODEL_NAME.value, ""), + ), + None, + ) + @staticmethod def _get_input_sequence_length( standard_logging_payload: StandardLoggingPayload, diff --git a/litellm/types/integrations/prometheus.py b/litellm/types/integrations/prometheus.py index 8f4ad26a4fa..8b94edb1901 100644 --- a/litellm/types/integrations/prometheus.py +++ b/litellm/types/integrations/prometheus.py @@ -293,6 +293,10 @@ DEFINED_PROMETHEUS_METRICS = Literal[ "litellm_deployment_rpm_limit", "litellm_remaining_api_key_requests_for_model", "litellm_remaining_api_key_tokens_for_model", + "litellm_remaining_team_requests_for_model", + "litellm_remaining_team_tokens_for_model", + "litellm_team_rpm_limit", + "litellm_team_tpm_limit", "litellm_api_key_rate_limit_allowed_metric", "litellm_api_key_rate_limit_used_metric", "litellm_team_rate_limit_allowed_metric", @@ -821,6 +825,17 @@ class PrometheusMetricLabels: UserAPIKeyLabelNames.MODEL_ID.value, ] + litellm_remaining_team_requests_for_model: ClassVar[Sequence[str]] = [ + UserAPIKeyLabelNames.TEAM.value, + UserAPIKeyLabelNames.TEAM_ALIAS.value, + UserAPIKeyLabelNames.v1_LITELLM_MODEL_NAME.value, + ] + + litellm_remaining_team_tokens_for_model = litellm_remaining_team_requests_for_model + + litellm_team_rpm_limit = litellm_remaining_team_requests_for_model + + litellm_team_tpm_limit = litellm_remaining_team_requests_for_model litellm_api_key_rate_limit_allowed_metric: ClassVar[tuple[str, ...]] = ( UserAPIKeyLabelNames.API_KEY_HASH.value, UserAPIKeyLabelNames.API_KEY_ALIAS.value, @@ -1152,6 +1167,9 @@ class NoOpMetric: def labels(self, *args, **kwargs): return self + def remove(self, *labelvalues: object) -> None: + pass + def inc(self, *args, **kwargs) -> None: pass diff --git a/tests/unit/integrations/test_prometheus_client_ip_user_agent.py b/tests/unit/integrations/test_prometheus_client_ip_user_agent.py index 004ac4dbffb..aec46d101da 100644 --- a/tests/unit/integrations/test_prometheus_client_ip_user_agent.py +++ b/tests/unit/integrations/test_prometheus_client_ip_user_agent.py @@ -94,6 +94,7 @@ async def test_async_post_call_success_hook_includes_client_ip_user_agent(): logger._increment_token_metrics = MagicMock() logger._increment_remaining_budget_metrics = AsyncMock() logger._set_virtual_key_rate_limit_metrics = MagicMock() + logger._set_team_rate_limit_metrics = MagicMock() logger._set_key_and_team_rate_limit_metrics = MagicMock() logger._set_latency_metrics = MagicMock() logger.set_llm_deployment_success_metrics = MagicMock() diff --git a/tests/unit/integrations/test_prometheus_team_rate_limit_metrics.py b/tests/unit/integrations/test_prometheus_team_rate_limit_metrics.py new file mode 100644 index 00000000000..687d272f60e --- /dev/null +++ b/tests/unit/integrations/test_prometheus_team_rate_limit_metrics.py @@ -0,0 +1,549 @@ +""" +Tests for the team-scoped rate limit Prometheus gauges. + +LiteLLM exposed configured/remaining rate limits at virtual key scope +(``litellm_remaining_api_key_*_for_model``) and deployment scope +(``litellm_deployment_{tpm,rpm}_limit``) but not at team scope, so there was +no way to alert on a team approaching the ``model_tpm_limit`` / +``model_rpm_limit`` configured on its team object. + +The v3 rate limiter already computes those numbers for its ``model_per_team`` +descriptor and ships them to clients as +``x-ratelimit-model_per_team-{remaining,limit}-{requests,tokens}``. These +tests cover routing those already-computed values to Prometheus. +""" + +from typing import get_args +from unittest.mock import MagicMock, patch + +import pytest + +from prometheus_client import CollectorRegistry, Gauge + +from litellm.integrations.prometheus import PrometheusLogger, _ExcludedLabelMetric +from litellm.integrations.prometheus_helpers.bounded_prometheus_series_tracker import ( + BoundedPrometheusSeriesTracker, +) +from litellm.proxy.hooks.parallel_request_limiter_v3 import ( + _PROXY_MaxParallelRequestsHandler_v3, +) +from litellm.types.integrations.prometheus import ( + DEFINED_PROMETHEUS_METRICS, + NoOpMetric, + PrometheusMetricLabels, + UserAPIKeyLabelNames, +) + +TEAM_LABELS = {"team": "team-abc", "team_alias": "research", "model": "gpt-4o-mini"} +ORIGINAL_LABELNAMES = ("team", "team_alias", "model") + +TEAM_RATE_LIMIT_METRICS = ( + "litellm_remaining_team_requests_for_model", + "litellm_remaining_team_tokens_for_model", + "litellm_team_rpm_limit", + "litellm_team_tpm_limit", +) + + +@pytest.fixture(autouse=True) +def _single_process_collection(monkeypatch): + """ + Series retirement is only possible outside multiprocess collection, so pin + the mode rather than depending on whatever the ambient environment has set. + """ + monkeypatch.delenv("PROMETHEUS_MULTIPROC_DIR", raising=False) + monkeypatch.delenv("prometheus_multiproc_dir", raising=False) + + +def _logger_with_mock_team_gauges() -> PrometheusLogger: + with patch( # test-quality-ok: PrometheusLogger has no registry-injection seam; every test in this directory skips its metric construction the same way + "litellm.integrations.prometheus.PrometheusLogger.__init__", return_value=None + ): + logger = PrometheusLogger() + for metric_name in TEAM_RATE_LIMIT_METRICS: + setattr(logger, metric_name, MagicMock()) + logger.get_labels_for_metric = MagicMock(side_effect=PrometheusMetricLabels.get_labels) + logger._team_series_label_values = {} + # the real tracker, not a double: it owns the removal these tests assert on + logger._bounded_prometheus_series_tracker = BoundedPrometheusSeriesTracker() + return logger + + +def _payload_with_headers(additional_headers: dict) -> dict: + return { + "metadata": {}, + "hidden_params": {"additional_headers": additional_headers}, + } + + +def _set_team_metrics(logger: PrometheusLogger, standard_logging_payload: dict) -> None: + logger._set_team_rate_limit_metrics( + user_api_team="team-abc", + user_api_team_alias="research", + model_group="gpt-4o-mini", + standard_logging_payload=standard_logging_payload, + ) + + +def _assert_set_once(logger: PrometheusLogger, metric_name: str, value: int) -> None: + getattr(logger, metric_name).labels.return_value.set.assert_called_once_with(value) + + +ALL_TEAM_HEADERS = { + "x-ratelimit-model_per_team-remaining-requests": 42, + "x-ratelimit-model_per_team-remaining-tokens": 900, + "x-ratelimit-model_per_team-limit-requests": 100, + "x-ratelimit-model_per_team-limit-tokens": 1000, +} + + +def test_team_metrics_are_defined_with_team_and_model_labels(): + defined_metrics = get_args(DEFINED_PROMETHEUS_METRICS) + expected_labels = [ + UserAPIKeyLabelNames.TEAM.value, + UserAPIKeyLabelNames.TEAM_ALIAS.value, + UserAPIKeyLabelNames.v1_LITELLM_MODEL_NAME.value, + ] + + for metric_name in TEAM_RATE_LIMIT_METRICS: + assert metric_name in defined_metrics + labels = PrometheusMetricLabels.get_labels(metric_name) + for expected_label in expected_labels: + assert expected_label in labels + + +def test_every_logger_owned_metric_resolves_labels(): + """ + ``PrometheusMetricLabels.get_labels`` resolves a metric name to a label + list via ``getattr``, so a metric added to the literal without a matching + label attribute fails at logger construction time in production rather + than at lint time. + + ``litellm_in_flight_requests`` is excluded because it is a label-free + gauge registered by the in-flight middleware, not by ``PrometheusLogger``; + it appears in the literal only so ``prometheus_metrics_config`` can name it. + """ + for metric_name in get_args(DEFINED_PROMETHEUS_METRICS): + if metric_name == "litellm_in_flight_requests": + continue + assert isinstance(PrometheusMetricLabels.get_labels(metric_name), list) + + +def test_sets_every_team_gauge_from_v3_headers(): + logger = _logger_with_mock_team_gauges() + + _set_team_metrics(logger, _payload_with_headers(dict(ALL_TEAM_HEADERS))) + + _assert_set_once(logger, "litellm_remaining_team_requests_for_model", 42) + _assert_set_once(logger, "litellm_remaining_team_tokens_for_model", 900) + _assert_set_once(logger, "litellm_team_rpm_limit", 100) + _assert_set_once(logger, "litellm_team_tpm_limit", 1000) + + +def test_labels_carry_team_and_requested_model(): + logger = _logger_with_mock_team_gauges() + + _set_team_metrics(logger, _payload_with_headers(dict(ALL_TEAM_HEADERS))) + + for metric_name in TEAM_RATE_LIMIT_METRICS: + labelnames = PrometheusMetricLabels.get_labels(metric_name) + label_values = getattr(logger, metric_name).labels.call_args.args + assert dict(zip(labelnames, label_values, strict=True)) == TEAM_LABELS + + +def test_emits_nothing_when_team_has_no_configured_limits(): + """A team without per-model limits gets no descriptor, so no header, so no series.""" + logger = _logger_with_mock_team_gauges() + + _set_team_metrics( + logger, + _payload_with_headers( + { + "x-ratelimit-model_per_key-remaining-requests": 42, + "x-ratelimit-model_per_key-limit-requests": 100, + } + ), + ) + + for metric_name in TEAM_RATE_LIMIT_METRICS: + getattr(logger, metric_name).labels.assert_not_called() + + +def test_drops_stale_series_when_a_team_limit_is_removed(): + """ + Prometheus keeps a child series for the life of the process once emitted, + so a team whose limit is removed would otherwise keep publishing the last + values it saw and alerts would fire on a limit nobody enforces. + """ + logger = _logger_with_mock_team_gauges() + + _set_team_metrics(logger, _payload_with_headers(dict(ALL_TEAM_HEADERS))) + _assert_set_once(logger, "litellm_remaining_team_requests_for_model", 42) + + _set_team_metrics(logger, _payload_with_headers({})) + + for metric_name in TEAM_RATE_LIMIT_METRICS: + gauge = getattr(logger, metric_name) + gauge.remove.assert_called_once_with("team-abc", "research", "gpt-4o-mini") + + +def test_survives_removing_a_series_that_was_never_emitted(): + """The common case: a team that never had a limit for this model.""" + logger = _logger_with_mock_team_gauges() + for metric_name in TEAM_RATE_LIMIT_METRICS: + getattr(logger, metric_name).remove.side_effect = KeyError("not present") + + _set_team_metrics(logger, _payload_with_headers({})) + + for metric_name in TEAM_RATE_LIMIT_METRICS: + getattr(logger, metric_name).labels.assert_not_called() + + +def test_emits_only_the_dimension_the_team_configured(): + """A team with only an RPM limit must not get a fabricated TPM series.""" + logger = _logger_with_mock_team_gauges() + + _set_team_metrics( + logger, + _payload_with_headers( + { + "x-ratelimit-model_per_team-remaining-requests": 7, + "x-ratelimit-model_per_team-limit-requests": 60, + } + ), + ) + + _assert_set_once(logger, "litellm_remaining_team_requests_for_model", 7) + _assert_set_once(logger, "litellm_team_rpm_limit", 60) + logger.litellm_remaining_team_tokens_for_model.labels.assert_not_called() + logger.litellm_team_tpm_limit.labels.assert_not_called() + + +def test_emits_zero_remaining_rather_than_skipping_it(): + """An exhausted team is the case operators alert on, so 0 must be a real sample.""" + logger = _logger_with_mock_team_gauges() + + _set_team_metrics( + logger, + _payload_with_headers( + { + "x-ratelimit-model_per_team-remaining-requests": 0, + "x-ratelimit-model_per_team-remaining-tokens": 0, + } + ), + ) + + _assert_set_once(logger, "litellm_remaining_team_requests_for_model", 0) + _assert_set_once(logger, "litellm_remaining_team_tokens_for_model", 0) + + +def test_emits_nothing_for_a_request_with_no_team(): + logger = _logger_with_mock_team_gauges() + + logger._set_team_rate_limit_metrics( + user_api_team=None, + user_api_team_alias=None, + model_group="gpt-4o-mini", + standard_logging_payload=_payload_with_headers(dict(ALL_TEAM_HEADERS)), + ) + + for metric_name in TEAM_RATE_LIMIT_METRICS: + getattr(logger, metric_name).labels.assert_not_called() + + +@pytest.mark.parametrize("bad_value", ["100", None, True, 12.5]) +def test_ignores_non_int_header_values(bad_value): + logger = _logger_with_mock_team_gauges() + + _set_team_metrics( + logger, + _payload_with_headers({"x-ratelimit-model_per_team-remaining-requests": bad_value}), + ) + + logger.litellm_remaining_team_requests_for_model.labels.assert_not_called() + + +def test_raises_nothing_when_payload_has_no_hidden_params(): + logger = _logger_with_mock_team_gauges() + + _set_team_metrics(logger, {"metadata": {}}) + + for metric_name in TEAM_RATE_LIMIT_METRICS: + getattr(logger, metric_name).labels.assert_not_called() + + +def test_limiter_publishes_team_headers_in_the_shape_the_gauges_read(): + """ + Pins the producer/consumer contract: the gauges read header names the v3 + limiter builds from ``descriptor_key`` + ``rate_limit_type``, so a change + to that format would otherwise silently stop the team series. + """ + headers = _PROXY_MaxParallelRequestsHandler_v3._merge_ratelimit_statuses_into_additional_headers( + additional_headers={}, + statuses=[ + { + "code": "OK", + "current_limit": 100, + "limit_remaining": 42, + "rate_limit_type": "requests", + "descriptor_key": "model_per_team", + }, + { + "code": "OK", + "current_limit": 1000, + "limit_remaining": 900, + "rate_limit_type": "tokens", + "descriptor_key": "model_per_team", + }, + ], + ) + + assert headers == { + "x-ratelimit-model_per_team-remaining-requests": 42, + "x-ratelimit-model_per_team-limit-requests": 100, + "x-ratelimit-model_per_team-remaining-tokens": 900, + "x-ratelimit-model_per_team-limit-tokens": 1000, + } + + +def _logger_with_real_gauge(metric_name: str, gauge: Gauge) -> PrometheusLogger: + logger = _logger_with_mock_team_gauges() + setattr(logger, metric_name, gauge) + return logger + + +def test_removes_a_real_prometheus_child_series_when_the_limit_disappears(): + """ + Mock gauges cannot prove the drop works, since prometheus_client owns the + child-series bookkeeping. This drives the real Gauge end to end. + """ + registry = CollectorRegistry() + gauge = Gauge("litellm_team_rpm_limit", "doc", labelnames=list(ORIGINAL_LABELNAMES), registry=registry) + logger = _logger_with_real_gauge("litellm_team_rpm_limit", gauge) + + _set_team_metrics(logger, _payload_with_headers({"x-ratelimit-model_per_team-limit-requests": 60})) + assert registry.get_sample_value("litellm_team_rpm_limit", TEAM_LABELS) == 60 + + _set_team_metrics(logger, _payload_with_headers({})) + assert registry.get_sample_value("litellm_team_rpm_limit", TEAM_LABELS) is None + + +def test_removing_a_never_emitted_real_series_raises_nothing(): + registry = CollectorRegistry() + gauge = Gauge("litellm_team_tpm_limit", "doc", labelnames=list(ORIGINAL_LABELNAMES), registry=registry) + logger = _logger_with_real_gauge("litellm_team_tpm_limit", gauge) + + _set_team_metrics(logger, _payload_with_headers({})) + + assert registry.get_sample_value("litellm_team_tpm_limit", TEAM_LABELS) is None + + +def test_excluded_label_wrapper_sets_and_removes_using_the_kept_labels(): + registry = CollectorRegistry() + real = Gauge("litellm_team_rpm_limit", "doc", labelnames=["team", "model"], registry=registry) + wrapper = _ExcludedLabelMetric(real, ORIGINAL_LABELNAMES, frozenset({"team_alias"})) + kept = {"team": "team-abc", "model": "gpt-4o-mini"} + + wrapper.labels(**TEAM_LABELS).set(60) + assert registry.get_sample_value("litellm_team_rpm_limit", kept) == 60 + + wrapper.remove(*(TEAM_LABELS[name] for name in ORIGINAL_LABELNAMES)) + assert registry.get_sample_value("litellm_team_rpm_limit", kept) is None + + +def test_excluded_label_wrapper_cannot_remove_when_every_label_is_excluded(): + """ + With every label excluded the gauge collapses to a single unlabeled sample, + which has no child series for prometheus_client to remove. Such a metric + cannot represent per-team state at all, so there is no correct value to + drop it to; this pins the behaviour rather than papering over it. + """ + registry = CollectorRegistry() + real = Gauge("litellm_team_rpm_limit", "doc", registry=registry) + wrapper = _ExcludedLabelMetric(real, ORIGINAL_LABELNAMES, frozenset(ORIGINAL_LABELNAMES)) + + wrapper.labels(**TEAM_LABELS).set(60) + assert registry.get_sample_value("litellm_team_rpm_limit", {}) == 60 + + wrapper.remove(*(TEAM_LABELS[name] for name in ORIGINAL_LABELNAMES)) + assert registry.get_sample_value("litellm_team_rpm_limit", {}) == 60 + + +def test_noop_metric_remove_is_inert(): + """A disabled metric answers every call without recording or raising.""" + metric = NoOpMetric() + + child = metric.labels(*TEAM_LABELS.values()) + + assert child is metric + assert child.set(60) is None + assert metric.remove(*TEAM_LABELS.values()) is None + + +def test_retires_the_old_series_when_a_team_is_renamed(): + """ + A rename changes team_alias, which starts a new series. The old one would + otherwise keep publishing the values it held at rename time, double + counting the team on any sum over `team`. + """ + registry = CollectorRegistry() + gauge = Gauge("litellm_team_rpm_limit", "doc", labelnames=list(ORIGINAL_LABELNAMES), registry=registry) + logger = _logger_with_real_gauge("litellm_team_rpm_limit", gauge) + headers = {"x-ratelimit-model_per_team-limit-requests": 60} + + _set_team_metrics(logger, _payload_with_headers(headers)) + assert registry.get_sample_value("litellm_team_rpm_limit", TEAM_LABELS) == 60 + + logger._set_team_rate_limit_metrics( + user_api_team="team-abc", + user_api_team_alias="ml-research", + model_group="gpt-4o-mini", + standard_logging_payload=_payload_with_headers(headers), + ) + + renamed = {**TEAM_LABELS, "team_alias": "ml-research"} + assert registry.get_sample_value("litellm_team_rpm_limit", renamed) == 60 + assert registry.get_sample_value("litellm_team_rpm_limit", TEAM_LABELS) is None + + +def test_keeps_other_teams_when_one_team_is_renamed(): + registry = CollectorRegistry() + gauge = Gauge("litellm_team_rpm_limit", "doc", labelnames=list(ORIGINAL_LABELNAMES), registry=registry) + logger = _logger_with_real_gauge("litellm_team_rpm_limit", gauge) + headers = {"x-ratelimit-model_per_team-limit-requests": 60} + + logger._set_team_rate_limit_metrics( + user_api_team="team-other", + user_api_team_alias="platform", + model_group="gpt-4o-mini", + standard_logging_payload=_payload_with_headers(headers), + ) + _set_team_metrics(logger, _payload_with_headers(headers)) + logger._set_team_rate_limit_metrics( + user_api_team="team-abc", + user_api_team_alias="ml-research", + model_group="gpt-4o-mini", + standard_logging_payload=_payload_with_headers(headers), + ) + + other = {"team": "team-other", "team_alias": "platform", "model": "gpt-4o-mini"} + assert registry.get_sample_value("litellm_team_rpm_limit", other) == 60 + + +def test_emits_nothing_when_the_team_label_is_excluded(): + """ + Without a team label the gauge collapses to one sample shared by every + team, which attributes a limit to nobody and cannot be retired. + """ + logger = _logger_with_mock_team_gauges() + logger.get_labels_for_metric = MagicMock(return_value=["model"]) + + _set_team_metrics(logger, _payload_with_headers(dict(ALL_TEAM_HEADERS))) + + for metric_name in TEAM_RATE_LIMIT_METRICS: + getattr(logger, metric_name).labels.assert_not_called() + getattr(logger, metric_name).remove.assert_not_called() + + +def test_excluded_labels_never_reach_team_gauge_labelnames(): + """ + `exclude_labels` is applied inside `get_labels_for_metric`, so the + labelnames a team gauge is constructed with never contain an excluded + label. The factory only wraps a metric when its labelnames still intersect + `exclude_labels`, so these gauges are always real prometheus_client + Gauges and always expose `collect` for alias cleanup. + """ + with patch( # test-quality-ok: PrometheusLogger has no registry-injection seam; every test in this directory skips its metric construction the same way + "litellm.integrations.prometheus.PrometheusLogger.__init__", return_value=None + ): + logger = PrometheusLogger() + logger.exclude_labels = frozenset({"model", "team_alias"}) + logger.label_filters = {} + logger._cached_metric_labels = {} + + for metric_name in TEAM_RATE_LIMIT_METRICS: + labelnames = logger.get_labels_for_metric(metric_name) + assert not frozenset(labelnames) & logger.exclude_labels + assert "team" in labelnames + + +def test_retires_the_old_alias_when_the_limit_is_removed_after_a_rename(): + """ + A team can be renamed and then have its limit removed before it sends + another limited request. The removal path only knows the current + labelset, so without sweeping superseded aliases on that path too, the + pre-rename series would stay published forever. + """ + registry = CollectorRegistry() + gauge = Gauge("litellm_team_rpm_limit", "doc", labelnames=list(ORIGINAL_LABELNAMES), registry=registry) + logger = _logger_with_real_gauge("litellm_team_rpm_limit", gauge) + + _set_team_metrics(logger, _payload_with_headers({"x-ratelimit-model_per_team-limit-requests": 60})) + assert registry.get_sample_value("litellm_team_rpm_limit", TEAM_LABELS) == 60 + + logger._set_team_rate_limit_metrics( + user_api_team="team-abc", + user_api_team_alias="ml-research", + model_group="gpt-4o-mini", + standard_logging_payload=_payload_with_headers({}), + ) + + assert registry.get_sample_value("litellm_team_rpm_limit", TEAM_LABELS) is None + renamed = {**TEAM_LABELS, "team_alias": "ml-research"} + assert registry.get_sample_value("litellm_team_rpm_limit", renamed) is None + + +def test_rename_survives_a_tracked_series_that_is_already_gone(): + """ + The tracked labelset can outlive the child series it names, for instance + when a cardinality cap evicts it. Retiring it must not break the emission + that triggered the retirement. + """ + registry = CollectorRegistry() + gauge = Gauge("litellm_team_rpm_limit", "doc", labelnames=list(ORIGINAL_LABELNAMES), registry=registry) + logger = _logger_with_real_gauge("litellm_team_rpm_limit", gauge) + headers = {"x-ratelimit-model_per_team-limit-requests": 60} + + _set_team_metrics(logger, _payload_with_headers(headers)) + gauge.remove(*(TEAM_LABELS[name] for name in ORIGINAL_LABELNAMES)) + assert registry.get_sample_value("litellm_team_rpm_limit", TEAM_LABELS) is None + + logger._set_team_rate_limit_metrics( + user_api_team="team-abc", + user_api_team_alias="ml-research", + model_group="gpt-4o-mini", + standard_logging_payload=_payload_with_headers(headers), + ) + + renamed = {**TEAM_LABELS, "team_alias": "ml-research"} + assert registry.get_sample_value("litellm_team_rpm_limit", renamed) == 60 + + +def test_does_not_attempt_retirement_under_multiprocess_collection(monkeypatch): + """ + prometheus_client refuses to remove a labelset when PROMETHEUS_MULTIPROC_DIR + is set, warning instead, because a worker cannot retire a series another + worker wrote. Attempting it on every team request would emit warnings while + leaving the sample in place, so the gauges are set and nothing is retired. + """ + monkeypatch.setenv("PROMETHEUS_MULTIPROC_DIR", "/tmp/does-not-need-to-exist") + logger = _logger_with_mock_team_gauges() + + _set_team_metrics(logger, _payload_with_headers(dict(ALL_TEAM_HEADERS))) + _set_team_metrics(logger, _payload_with_headers({})) + + _assert_set_once(logger, "litellm_remaining_team_requests_for_model", 42) + for metric_name in TEAM_RATE_LIMIT_METRICS: + getattr(logger, metric_name).remove.assert_not_called() + + +def test_retires_series_when_collection_is_single_process(monkeypatch): + monkeypatch.delenv("PROMETHEUS_MULTIPROC_DIR", raising=False) + monkeypatch.delenv("prometheus_multiproc_dir", raising=False) + logger = _logger_with_mock_team_gauges() + + _set_team_metrics(logger, _payload_with_headers(dict(ALL_TEAM_HEADERS))) + _set_team_metrics(logger, _payload_with_headers({})) + + for metric_name in TEAM_RATE_LIMIT_METRICS: + getattr(logger, metric_name).remove.assert_called_once_with("team-abc", "research", "gpt-4o-mini")