From ec4e0146c7a32da649011e5ab27bf750988dd805 Mon Sep 17 00:00:00 2001 From: yucheng-berri Date: Fri, 26 Jun 2026 15:26:55 -0700 Subject: [PATCH] feat(prometheus): add requested_model label to spend and requests metrics (#31410) litellm_spend_metric_total and litellm_requests_metric_total previously exposed only the resolved backend model_id and friendly model name, so operators could not group spend or request counts by the model alias the caller actually asked for when a router fronts multiple deployments behind one name. This adds the existing UserAPIKeyLabelNames.REQUESTED_MODEL to both labelname lists; the value is already populated upstream from standard_logging_payload["model_group"] and flows through the shared _increment_top_level_request_and_spend_metrics call site. The sibling token metrics (input/output/total) already carry the label, so this also restores cross-metric consistency. Resolves LIT-3796 --- litellm/types/integrations/prometheus.py | 2 ++ .../test_prometheus_logging_callbacks.py | 2 ++ .../integrations/test_prometheus_labels.py | 28 +++++++++++++++++++ 3 files changed, 32 insertions(+) diff --git a/litellm/types/integrations/prometheus.py b/litellm/types/integrations/prometheus.py index d203fe73d02..ba427de41a2 100644 --- a/litellm/types/integrations/prometheus.py +++ b/litellm/types/integrations/prometheus.py @@ -409,6 +409,7 @@ class PrometheusMetricLabels: UserAPIKeyLabelNames.USER_EMAIL.value, UserAPIKeyLabelNames.CLIENT_IP.value, UserAPIKeyLabelNames.USER_AGENT.value, + UserAPIKeyLabelNames.REQUESTED_MODEL.value, UserAPIKeyLabelNames.MODEL_ID.value, UserAPIKeyLabelNames.API_PROVIDER.value, ] @@ -424,6 +425,7 @@ class PrometheusMetricLabels: UserAPIKeyLabelNames.USER_EMAIL.value, UserAPIKeyLabelNames.CLIENT_IP.value, UserAPIKeyLabelNames.USER_AGENT.value, + UserAPIKeyLabelNames.REQUESTED_MODEL.value, UserAPIKeyLabelNames.MODEL_ID.value, UserAPIKeyLabelNames.API_PROVIDER.value, ] diff --git a/tests/enterprise/litellm_enterprise/enterprise_callbacks/test_prometheus_logging_callbacks.py b/tests/enterprise/litellm_enterprise/enterprise_callbacks/test_prometheus_logging_callbacks.py index d0ad1cc8f82..be199c19149 100644 --- a/tests/enterprise/litellm_enterprise/enterprise_callbacks/test_prometheus_logging_callbacks.py +++ b/tests/enterprise/litellm_enterprise/enterprise_callbacks/test_prometheus_logging_callbacks.py @@ -607,6 +607,7 @@ def test_increment_top_level_request_and_spend_metrics(prometheus_logger): api_provider="openai", client_ip=None, user_agent=None, + requested_model=None, ) prometheus_logger.litellm_requests_metric.labels().inc.assert_called_once() @@ -626,6 +627,7 @@ def test_increment_top_level_request_and_spend_metrics(prometheus_logger): api_provider="openai", client_ip=None, user_agent=None, + requested_model=None, ) prometheus_logger.litellm_spend_metric.labels().inc.assert_called_once_with(0.1) diff --git a/tests/test_litellm/integrations/test_prometheus_labels.py b/tests/test_litellm/integrations/test_prometheus_labels.py index c83d89e87c4..70cc9ab33d6 100644 --- a/tests/test_litellm/integrations/test_prometheus_labels.py +++ b/tests/test_litellm/integrations/test_prometheus_labels.py @@ -148,6 +148,34 @@ def test_model_id_in_required_metrics(): print(f"✅ {metric_name} contains model_id label") +def test_requested_model_in_spend_and_requests_metrics(): + """ + Regression test for LIT-3796. + + litellm_spend_metric and litellm_requests_metric must expose the + requested_model label so spend and request counts can be grouped by the + model alias the caller asked for, not just the backend deployment that + served the request. The sibling token metrics (input/output/total) + already carry this label; spend and requests were the odd ones out, even + though both are emitted side-by-side from the same call site. + """ + requested_model_label = UserAPIKeyLabelNames.REQUESTED_MODEL.value + + metrics_with_requested_model = [ + "litellm_spend_metric", + "litellm_requests_metric", + "litellm_input_tokens_metric", + "litellm_output_tokens_metric", + "litellm_total_tokens_metric", + ] + + for metric_name in metrics_with_requested_model: + labels = PrometheusMetricLabels.get_labels(metric_name) + assert requested_model_label in labels, ( + f"Metric {metric_name} should contain requested_model label" + ) + + def test_route_normalization_for_responses_api(): """ Test that route normalization prevents high cardinality in Prometheus metrics