From 6aac068555a017248ac98d9fa0a04881663c4be0 Mon Sep 17 00:00:00 2001 From: DanBrima <40828002+DanBrima@users.noreply.github.com> Date: Tue, 18 Aug 2026 05:22:58 +0000 Subject: [PATCH] fix(prometheus): retire superseded team rate limit series Two ways a team gauge could keep publishing values nobody enforces. Renaming a team changes the team_alias label, which starts a new child series and leaves the old one holding the values it had at rename time, double counting the team on any sum over team. Before setting a series, retire any series for the same team and model under a different alias. Excluding the team label collapses the gauge to a single sample shared by every team, attributing a limit to nobody and leaving nothing that can be retired. Emit nothing in that configuration rather than a number that silently belongs to whichever team wrote it last. Team tests now resolve real label sets instead of an empty list, which is what the metrics are actually constructed with. --- litellm/integrations/prometheus.py | 47 ++++++++++- ...test_prometheus_team_rate_limit_metrics.py | 78 +++++++++++++++++-- 2 files changed, 115 insertions(+), 10 deletions(-) diff --git a/litellm/integrations/prometheus.py b/litellm/integrations/prometheus.py index 848dba2d32a..ac008e3714a 100644 --- a/litellm/integrations/prometheus.py +++ b/litellm/integrations/prometheus.py @@ -2122,22 +2122,65 @@ class PrometheusLogger(CustomLogger): values it ever saw, and alerts would evaluate against a number no longer being enforced. """ + labelnames: Final = self.get_labels_for_metric(metric_name) + if UserAPIKeyLabelNames.TEAM.value not in labelnames: + # Without a team label the gauge collapses to one sample shared by + # every team, which cannot attribute a limit to anyone and cannot + # be retired. Publishing nothing beats publishing a number that + # silently belongs to whichever team wrote it last. + return + labels: Final = prometheus_label_factory( - supported_enum_labels=self.get_labels_for_metric(metric_name), + supported_enum_labels=labelnames, enum_values=enum_values, label_context=label_context, ) if value is not None: + self._drop_superseded_team_series(gauge=gauge, labelnames=labelnames, labels=labels) gauge.labels(**labels).set(value) return try: - gauge.remove(*(labels[name] for name in self.get_labels_for_metric(metric_name))) + gauge.remove(*(labels.get(name, "") for name in labelnames)) except KeyError: # No child series for this labelset, which is the common case: # the team never had a limit for this model. pass + def _drop_superseded_team_series( + self, + gauge: _LabeledGauge, + labelnames: Sequence[str], + labels: Mapping[str, str], + ) -> None: + """ + Retire child series that describe this same team and model under a + different alias. Renaming a team changes ``team_alias``, which starts a + new series, and the old one would otherwise keep publishing the values + it held at rename time, double counting the team on any sum over + ``team``. + """ + collect: Final = getattr(gauge, "collect", None) + if collect is None: + return + + team_label: Final = UserAPIKeyLabelNames.TEAM.value + alias_label: Final = UserAPIKeyLabelNames.TEAM_ALIAS.value + model_label: Final = UserAPIKeyLabelNames.v1_LITELLM_MODEL_NAME.value + superseded: Final = tuple( + tuple(sample.labels.get(name, "") for name in labelnames) + for metric in collect() + for sample in metric.samples + if sample.labels.get(team_label) == labels.get(team_label) + and sample.labels.get(model_label) == labels.get(model_label) + and sample.labels.get(alias_label) != labels.get(alias_label) + ) + for label_values in superseded: + try: + gauge.remove(*label_values) + except KeyError: + pass + def _set_latency_metrics( self, kwargs: dict, diff --git a/tests/test_litellm/integrations/test_prometheus_team_rate_limit_metrics.py b/tests/test_litellm/integrations/test_prometheus_team_rate_limit_metrics.py index 13688e97611..041ad02c1c1 100644 --- a/tests/test_litellm/integrations/test_prometheus_team_rate_limit_metrics.py +++ b/tests/test_litellm/integrations/test_prometheus_team_rate_limit_metrics.py @@ -42,15 +42,12 @@ TEAM_RATE_LIMIT_METRICS = ( ) -def _logger_with_mock_team_gauges(labels_are_real: bool = False) -> PrometheusLogger: +def _logger_with_mock_team_gauges() -> PrometheusLogger: with patch("litellm.integrations.prometheus.PrometheusLogger.__init__", return_value=None): logger = PrometheusLogger() for metric_name in TEAM_RATE_LIMIT_METRICS: setattr(logger, metric_name, MagicMock()) - if labels_are_real: - logger.get_labels_for_metric = MagicMock(side_effect=PrometheusMetricLabels.get_labels) - else: - logger.get_labels_for_metric = MagicMock(return_value=[]) + logger.get_labels_for_metric = MagicMock(side_effect=PrometheusMetricLabels.get_labels) return logger @@ -126,7 +123,7 @@ def test_sets_every_team_gauge_from_v3_headers(): def test_labels_carry_team_and_requested_model(): - logger = _logger_with_mock_team_gauges(labels_are_real=True) + logger = _logger_with_mock_team_gauges() _set_team_metrics(logger, _payload_with_headers(dict(ALL_TEAM_HEADERS))) @@ -161,7 +158,7 @@ def test_drops_stale_series_when_a_team_limit_is_removed(): so a team whose limit is removed would otherwise keep publishing the last values it saw and alerts would fire on a limit nobody enforces. """ - logger = _logger_with_mock_team_gauges(labels_are_real=True) + logger = _logger_with_mock_team_gauges() _set_team_metrics(logger, _payload_with_headers(dict(ALL_TEAM_HEADERS))) _assert_set_once(logger, "litellm_remaining_team_requests_for_model", 42) @@ -293,7 +290,7 @@ def test_limiter_publishes_team_headers_in_the_shape_the_gauges_read(): def _logger_with_real_gauge(metric_name: str, gauge: Gauge) -> PrometheusLogger: - logger = _logger_with_mock_team_gauges(labels_are_real=True) + logger = _logger_with_mock_team_gauges() setattr(logger, metric_name, gauge) return logger @@ -360,3 +357,68 @@ def test_noop_metric_remove_is_inert(): metric.labels(**TEAM_LABELS).set(60) metric.remove(*TEAM_LABELS.values()) + + +def test_retires_the_old_series_when_a_team_is_renamed(): + """ + A rename changes team_alias, which starts a new series. The old one would + otherwise keep publishing the values it held at rename time, double + counting the team on any sum over `team`. + """ + registry = CollectorRegistry() + gauge = Gauge("litellm_team_rpm_limit", "doc", labelnames=list(ORIGINAL_LABELNAMES), registry=registry) + logger = _logger_with_real_gauge("litellm_team_rpm_limit", gauge) + headers = {"x-ratelimit-model_per_team-limit-requests": 60} + + _set_team_metrics(logger, _payload_with_headers(headers)) + assert registry.get_sample_value("litellm_team_rpm_limit", TEAM_LABELS) == 60 + + logger._set_team_rate_limit_metrics( + user_api_team="team-abc", + user_api_team_alias="ml-research", + model_group="gpt-4o-mini", + standard_logging_payload=_payload_with_headers(headers), + ) + + renamed = {**TEAM_LABELS, "team_alias": "ml-research"} + assert registry.get_sample_value("litellm_team_rpm_limit", renamed) == 60 + assert registry.get_sample_value("litellm_team_rpm_limit", TEAM_LABELS) is None + + +def test_keeps_other_teams_when_one_team_is_renamed(): + registry = CollectorRegistry() + gauge = Gauge("litellm_team_rpm_limit", "doc", labelnames=list(ORIGINAL_LABELNAMES), registry=registry) + logger = _logger_with_real_gauge("litellm_team_rpm_limit", gauge) + headers = {"x-ratelimit-model_per_team-limit-requests": 60} + + logger._set_team_rate_limit_metrics( + user_api_team="team-other", + user_api_team_alias="platform", + model_group="gpt-4o-mini", + standard_logging_payload=_payload_with_headers(headers), + ) + _set_team_metrics(logger, _payload_with_headers(headers)) + logger._set_team_rate_limit_metrics( + user_api_team="team-abc", + user_api_team_alias="ml-research", + model_group="gpt-4o-mini", + standard_logging_payload=_payload_with_headers(headers), + ) + + other = {"team": "team-other", "team_alias": "platform", "model": "gpt-4o-mini"} + assert registry.get_sample_value("litellm_team_rpm_limit", other) == 60 + + +def test_emits_nothing_when_the_team_label_is_excluded(): + """ + Without a team label the gauge collapses to one sample shared by every + team, which attributes a limit to nobody and cannot be retired. + """ + logger = _logger_with_mock_team_gauges() + logger.get_labels_for_metric = MagicMock(return_value=["model"]) + + _set_team_metrics(logger, _payload_with_headers(dict(ALL_TEAM_HEADERS))) + + for metric_name in TEAM_RATE_LIMIT_METRICS: + getattr(logger, metric_name).labels.assert_not_called() + getattr(logger, metric_name).remove.assert_not_called()