From 9cb60c40cda1a7f5ec18541ec4038d995fcffbd0 Mon Sep 17 00:00:00 2001 From: DanBrima Date: Mon, 28 Sep 2026 16:00:32 +0300 Subject: [PATCH] fix(prometheus): require the model label before emitting a team gauge The guard only required the team label. These gauges are team-and-model scoped, so when `model` is excluded through prometheus label configuration every model for a team collapses onto one series: a request for a model with no configured limit retires the series a limited model published, and a second limited model overwrites it. Extends the existing team-label guard to require both labels, which is the same reasoning already applied to the team label -- a series that cannot attribute a limit to a specific team and model can neither be trusted nor retired. Reported by veria-ai on #37215. Co-Authored-By: Claude Opus 5 (1M context) --- litellm/integrations/prometheus.py | 18 +++++++++++++----- .../test_prometheus_team_rate_limit_metrics.py | 18 ++++++++++++++++++ 2 files changed, 31 insertions(+), 5 deletions(-) diff --git a/litellm/integrations/prometheus.py b/litellm/integrations/prometheus.py index f20921340c9..b186c23d51e 100644 --- a/litellm/integrations/prometheus.py +++ b/litellm/integrations/prometheus.py @@ -2393,11 +2393,19 @@ class PrometheusLogger(CustomLogger): removal path of this file's own. """ labelnames: Final = self.get_labels_for_metric(metric_name) - if UserAPIKeyLabelNames.TEAM.value not in labelnames: - # Without a team label the gauge collapses to one sample shared by - # every team, which cannot attribute a limit to anyone and cannot - # be retired. Publishing nothing beats publishing a number that - # silently belongs to whichever team wrote it last. + if ( + UserAPIKeyLabelNames.TEAM.value not in labelnames + or UserAPIKeyLabelNames.v1_LITELLM_MODEL_NAME.value not in labelnames + ): + # This gauge is team-and-model scoped, so it needs both labels to + # identify what it is reporting. Without the team label it collapses + # to one sample shared by every team; without the model label, to one + # sample shared by every model of a team -- and then a request for an + # unlimited model retires the series a limited one published, while a + # second limited model overwrites it. Either way the series cannot + # attribute a limit to anyone and cannot be retired, so publishing + # nothing beats publishing a number that silently belongs to whichever + # request wrote it last. return labels: Final = prometheus_label_factory( diff --git a/tests/unit/integrations/test_prometheus_team_rate_limit_metrics.py b/tests/unit/integrations/test_prometheus_team_rate_limit_metrics.py index 687d272f60e..1fee333817f 100644 --- a/tests/unit/integrations/test_prometheus_team_rate_limit_metrics.py +++ b/tests/unit/integrations/test_prometheus_team_rate_limit_metrics.py @@ -445,6 +445,24 @@ def test_emits_nothing_when_the_team_label_is_excluded(): getattr(logger, metric_name).remove.assert_not_called() +def test_emits_nothing_when_the_model_label_is_excluded(): + """ + These gauges are team-and-model scoped, so without a model label every + model for one team collapses to a single sample. A request for a model + with no limit would then retire the series a limited model published, and + a second limited model would overwrite it -- the same reasoning that drops + the gauge when the team label is missing. + """ + logger = _logger_with_mock_team_gauges() + logger.get_labels_for_metric = MagicMock(return_value=["team", "team_alias"]) + + _set_team_metrics(logger, _payload_with_headers(dict(ALL_TEAM_HEADERS))) + + for metric_name in TEAM_RATE_LIMIT_METRICS: + getattr(logger, metric_name).labels.assert_not_called() + getattr(logger, metric_name).remove.assert_not_called() + + def test_excluded_labels_never_reach_team_gauge_labelnames(): """ `exclude_labels` is applied inside `get_labels_for_metric`, so the