From 2c1847f8a262a1967de3db2dfc9c341286eb71e6 Mon Sep 17 00:00:00 2001 From: ahamedshaik16 Date: Wed, 7 Oct 2026 23:48:45 +0530 Subject: [PATCH] fix(prometheus): add model_group label to end-to-end latency metrics (#44860) litellm_llm_api_latency_metric, litellm_llm_api_time_to_first_token_metric, litellm_request_total_latency_metric, and litellm_deployment_latency_per_output_token previously carried requested_model/litellm_model_name/model_id but not model_group, so pooled-deployment latency couldn't be grouped by model pool on dashboards -- only the proxy-overhead-only metrics (litellm_overhead_latency_metric and friends) had model_group. All four metrics read enum_values.model_group through the existing prometheus_label_factory plumbing, so no new parameter threading was needed, just the label-list addition. Co-authored-by: ahamedshaik16 <24526479+ahamedshaik16@users.noreply.github.com> --- litellm/types/integrations/prometheus.py | 4 + .../test_prometheus_logging_callbacks.py | 4 + .../integrations/test_prometheus_labels.py | 187 ++++++++++++++++++ 3 files changed, 195 insertions(+) diff --git a/litellm/types/integrations/prometheus.py b/litellm/types/integrations/prometheus.py index dc552a730ea..a951381a2c9 100644 --- a/litellm/types/integrations/prometheus.py +++ b/litellm/types/integrations/prometheus.py @@ -416,6 +416,7 @@ def _resolve_deployment_and_latency_caller_identity_labels( class PrometheusMetricLabels: litellm_llm_api_latency_metric = [ + UserAPIKeyLabelNames.MODEL_GROUP.value, UserAPIKeyLabelNames.v1_LITELLM_MODEL_NAME.value, UserAPIKeyLabelNames.API_KEY_HASH.value, UserAPIKeyLabelNames.API_KEY_ALIAS.value, @@ -430,6 +431,7 @@ class PrometheusMetricLabels: ] litellm_llm_api_time_to_first_token_metric = [ + UserAPIKeyLabelNames.MODEL_GROUP.value, UserAPIKeyLabelNames.v1_LITELLM_MODEL_NAME.value, UserAPIKeyLabelNames.API_KEY_HASH.value, UserAPIKeyLabelNames.API_KEY_ALIAS.value, @@ -444,6 +446,7 @@ class PrometheusMetricLabels: ] litellm_request_total_latency_metric = [ + UserAPIKeyLabelNames.MODEL_GROUP.value, UserAPIKeyLabelNames.END_USER.value, UserAPIKeyLabelNames.API_KEY_HASH.value, UserAPIKeyLabelNames.API_KEY_ALIAS.value, @@ -516,6 +519,7 @@ class PrometheusMetricLabels: ] litellm_deployment_latency_per_output_token = [ + UserAPIKeyLabelNames.MODEL_GROUP.value, UserAPIKeyLabelNames.v2_LITELLM_MODEL_NAME.value, UserAPIKeyLabelNames.MODEL_ID.value, UserAPIKeyLabelNames.API_BASE.value, diff --git a/tests/unit/enterprise/enterprise_callbacks/test_prometheus_logging_callbacks.py b/tests/unit/enterprise/enterprise_callbacks/test_prometheus_logging_callbacks.py index f1c80bb11ea..78545d3fb62 100644 --- a/tests/unit/enterprise/enterprise_callbacks/test_prometheus_logging_callbacks.py +++ b/tests/unit/enterprise/enterprise_callbacks/test_prometheus_logging_callbacks.py @@ -433,6 +433,7 @@ def test_set_latency_metrics(prometheus_logger): team_alias="test_team_alias", org_id=None, org_alias=None, + model_group="openai-gpt", requested_model="openai-gpt", model="gpt-5-mini", model_id="model-123", @@ -453,6 +454,7 @@ def test_set_latency_metrics(prometheus_logger): team_alias="test_team_alias", org_id=None, org_alias=None, + model_group="openai-gpt", requested_model="openai-gpt", model="gpt-5-mini", model_id="model-123", @@ -473,6 +475,7 @@ def test_set_latency_metrics(prometheus_logger): team_alias="test_team_alias", org_id=None, org_alias=None, + model_group="openai-gpt", requested_model="openai-gpt", model="gpt-5-mini", model_id="model-123", @@ -1054,6 +1057,7 @@ def test_set_llm_deployment_success_metrics(prometheus_logger): # Verify latency per output token metric prometheus_logger.litellm_deployment_latency_per_output_token.labels.assert_called_once_with( + model_group="my_custom_model_group", litellm_model_name="gpt-5-mini", model_id="model-123", api_base="https://api.openai.com", diff --git a/tests/unit/integrations/test_prometheus_labels.py b/tests/unit/integrations/test_prometheus_labels.py index 8a1f5f5a0a8..ead0ce8ecff 100644 --- a/tests/unit/integrations/test_prometheus_labels.py +++ b/tests/unit/integrations/test_prometheus_labels.py @@ -980,6 +980,193 @@ def test_deployment_tpm_rpm_limit_metrics_emit_model_group_from_enum_values(): _clear_prometheus_registry() +def test_model_group_in_latency_metrics(): + """ + Test that model_group label is present on the end-to-end / per-call + latency metrics needed to build model-group latency dashboards. These + metrics previously only carried requested_model, litellm_model_name and + model_id, none of which identify the model_group a pooled deployment + belongs to -- only the proxy-overhead-only latency metrics + (litellm_overhead_latency_metric and friends) carried model_group. + """ + model_group_label = UserAPIKeyLabelNames.MODEL_GROUP.value + + metrics_with_model_group = [ + "litellm_llm_api_latency_metric", + "litellm_llm_api_time_to_first_token_metric", + "litellm_request_total_latency_metric", + "litellm_deployment_latency_per_output_token", + ] + + for metric_name in metrics_with_model_group: + labels = PrometheusMetricLabels.get_labels(metric_name) + assert ( + model_group_label in labels + ), f"Metric {metric_name} should contain model_group label" + print(f"✅ {metric_name} contains model_group label") + + +def test_model_group_value_flows_through_latency_metrics_label_factory(): + """ + The label being in the allow-list is necessary but not sufficient: the + factory must also carry the value from the enum through to the emitted + label. This would fail if the label were dropped from a metric's list or + if the value plumbing regressed, which the allow-list assertion above + cannot catch on its own. + """ + from unittest.mock import MagicMock + + from litellm.integrations.prometheus import ( + PrometheusLogger, + UserAPIKeyLabelValues, + prometheus_label_factory, + ) + + prometheus_logger = MagicMock() + prometheus_logger._cached_metric_labels = {} + prometheus_logger.label_filters = {} + prometheus_logger.get_labels_for_metric = ( + PrometheusLogger.get_labels_for_metric.__get__(prometheus_logger) + ) + + enum_values = UserAPIKeyLabelValues( + model_group="example-model-group", + litellm_model_name="gpt-4o-mini", + requested_model="example-model-group", + status_code="200", + ) + + for metric_name in [ + "litellm_llm_api_latency_metric", + "litellm_llm_api_time_to_first_token_metric", + "litellm_request_total_latency_metric", + "litellm_deployment_latency_per_output_token", + ]: + labels = prometheus_label_factory( + supported_enum_labels=prometheus_logger.get_labels_for_metric( + metric_name=metric_name + ), + enum_values=enum_values, + ) + assert ( + labels.get("model_group") == "example-model-group" + ), f"{metric_name} should emit model_group=example-model-group, got {labels.get('model_group')!r}" + + +def test_latency_metrics_emit_model_group_from_set_latency_metrics(): + """ + End-to-end emit wiring for _set_latency_metrics. + + The label-list and factory tests above prove the label exists and that + the factory carries a value handed to it, but neither drives the real + _set_latency_metrics code path, so deleting the production model_group + plumbing there would still pass them. This calls it directly with a + streaming request (so the time-to-first-token branch also fires) and + asserts the real litellm_llm_api_latency_metric, + litellm_llm_api_time_to_first_token_metric and + litellm_request_total_latency_metric Histogram series actually carry it. + """ + import datetime + + from litellm.integrations.prometheus import PrometheusLogger, UserAPIKeyLabelValues + + _clear_prometheus_registry() + try: + logger = PrometheusLogger() + start_time = datetime.datetime(2024, 1, 1, 0, 0, 0) + api_call_start_time = datetime.datetime(2024, 1, 1, 0, 0, 1) + completion_start_time = datetime.datetime(2024, 1, 1, 0, 0, 2) + end_time = datetime.datetime(2024, 1, 1, 0, 0, 3) + + enum_values = UserAPIKeyLabelValues( + model_group="example-model-group", + litellm_model_name="gpt-4o-mini", + requested_model="example-model-group", + status_code="200", + ) + + logger._set_latency_metrics( + kwargs={ + "start_time": start_time, + "end_time": end_time, + "api_call_start_time": api_call_start_time, + "completion_start_time": completion_start_time, + "stream": True, + "litellm_params": {"metadata": {}}, + }, + model="gpt-4o-mini", + user_api_key=None, + user_api_key_alias=None, + user_api_team=None, + user_api_team_alias=None, + enum_values=enum_values, + ) + + for metric in ( + logger.litellm_llm_api_latency_metric, + logger.litellm_llm_api_time_to_first_token_metric, + logger.litellm_request_total_latency_metric, + ): + index = metric._labelnames.index("model_group") + values = {sample_key[index] for sample_key in metric._metrics} + assert values == {"example-model-group"}, ( + f"expected model_group=example-model-group on {metric._name}, got {values}" + ) + finally: + _clear_prometheus_registry() + + +def test_deployment_latency_per_output_token_emits_model_group_from_enum_values(): + """ + End-to-end emit wiring for litellm_deployment_latency_per_output_token. + + Drives set_llm_deployment_success_metrics directly (its only caller) with + output_tokens > 0 so the latency-per-token branch fires, and asserts the + real Histogram series carries model_group; fails if that label-list + addition or the enum_values plumbing is removed. + """ + import datetime + + from litellm.integrations.prometheus import PrometheusLogger, UserAPIKeyLabelValues + + _clear_prometheus_registry() + try: + logger = PrometheusLogger() + start_time = datetime.datetime(2024, 1, 1, 0, 0, 0) + end_time = datetime.datetime(2024, 1, 1, 0, 0, 2) + enum_values = UserAPIKeyLabelValues( + model_group="example-model-group", + litellm_model_name="gpt-4o-mini", + requested_model="example-model-group", + status_code="200", + ) + logger.set_llm_deployment_success_metrics( + request_kwargs={ + "model": "gpt-4o-mini", + "litellm_params": {"metadata": {"model_info": {"id": "model-123"}}}, + "standard_logging_object": { + "model_group": "example-model-group", + "model_id": "model-123", + "api_base": "https://api.openai.com", + "hidden_params": {"additional_headers": None, "litellm_overhead_time_ms": None}, + }, + }, + start_time=start_time, + end_time=end_time, + enum_values=enum_values, + output_tokens=10.0, + ) + + metric = logger.litellm_deployment_latency_per_output_token + index = metric._labelnames.index("model_group") + values = {sample_key[index] for sample_key in metric._metrics} + assert values == {"example-model-group"}, ( + f"expected model_group=example-model-group on {metric._name}, got {values}" + ) + finally: + _clear_prometheus_registry() + + if __name__ == "__main__": test_user_email_in_required_metrics() test_user_email_label_exists()