diff --git a/litellm/integrations/prometheus.py b/litellm/integrations/prometheus.py index f20921340c9..b186c23d51e 100644 --- a/litellm/integrations/prometheus.py +++ b/litellm/integrations/prometheus.py @@ -2393,11 +2393,19 @@ class PrometheusLogger(CustomLogger): removal path of this file's own. """ labelnames: Final = self.get_labels_for_metric(metric_name) - if UserAPIKeyLabelNames.TEAM.value not in labelnames: - # Without a team label the gauge collapses to one sample shared by - # every team, which cannot attribute a limit to anyone and cannot - # be retired. Publishing nothing beats publishing a number that - # silently belongs to whichever team wrote it last. + if ( + UserAPIKeyLabelNames.TEAM.value not in labelnames + or UserAPIKeyLabelNames.v1_LITELLM_MODEL_NAME.value not in labelnames + ): + # This gauge is team-and-model scoped, so it needs both labels to + # identify what it is reporting. Without the team label it collapses + # to one sample shared by every team; without the model label, to one + # sample shared by every model of a team -- and then a request for an + # unlimited model retires the series a limited one published, while a + # second limited model overwrites it. Either way the series cannot + # attribute a limit to anyone and cannot be retired, so publishing + # nothing beats publishing a number that silently belongs to whichever + # request wrote it last. return labels: Final = prometheus_label_factory( diff --git a/tests/unit/integrations/test_prometheus_team_rate_limit_metrics.py b/tests/unit/integrations/test_prometheus_team_rate_limit_metrics.py index 687d272f60e..1fee333817f 100644 --- a/tests/unit/integrations/test_prometheus_team_rate_limit_metrics.py +++ b/tests/unit/integrations/test_prometheus_team_rate_limit_metrics.py @@ -445,6 +445,24 @@ def test_emits_nothing_when_the_team_label_is_excluded(): getattr(logger, metric_name).remove.assert_not_called() +def test_emits_nothing_when_the_model_label_is_excluded(): + """ + These gauges are team-and-model scoped, so without a model label every + model for one team collapses to a single sample. A request for a model + with no limit would then retire the series a limited model published, and + a second limited model would overwrite it -- the same reasoning that drops + the gauge when the team label is missing. + """ + logger = _logger_with_mock_team_gauges() + logger.get_labels_for_metric = MagicMock(return_value=["team", "team_alias"]) + + _set_team_metrics(logger, _payload_with_headers(dict(ALL_TEAM_HEADERS))) + + for metric_name in TEAM_RATE_LIMIT_METRICS: + getattr(logger, metric_name).labels.assert_not_called() + getattr(logger, metric_name).remove.assert_not_called() + + def test_excluded_labels_never_reach_team_gauge_labelnames(): """ `exclude_labels` is applied inside `get_labels_for_metric`, so the