From 0590b1eb3a1a91c2b9871cd1952892426b957513 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Tue, 27 May 2025 17:06:58 -0700 Subject: [PATCH] [Fix] Prometheus Metrics - Do not track end_user by default + expose flag to enable tracking end_user on prometheus (#11192) * fix: testing for disabling end user on metrics * fix: fixes for test_prometheus_factory * Delete litellm/model_prices_and_context_window_backup.json * fix: issues with merge conflicts * fix: test_get_end_user_id_for_cost_tracking_prometheus_only * Update tests/test_litellm/integrations/test_prometheus.py Co-authored-by: Copilot <175728472+Copilot@users.noreply.github.com> --------- Co-authored-by: Copilot <175728472+Copilot@users.noreply.github.com> --- litellm/__init__.py | 1 + ...odel_prices_and_context_window_backup.json | 128 ---------------- litellm/utils.py | 13 +- tests/litellm_utils_tests/test_utils.py | 14 +- .../test_prometheus_unit_tests.py | 20 +-- .../integrations/test_prometheus.py | 144 +++++++++++++++++- 6 files changed, 168 insertions(+), 152 deletions(-) diff --git a/litellm/__init__.py b/litellm/__init__.py index 9b019daec76..787c553b6ea 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -296,6 +296,7 @@ tag_budget_config: Optional[Dict[str, BudgetConfig]] = None max_end_user_budget: Optional[float] = None disable_end_user_cost_tracking: Optional[bool] = None disable_end_user_cost_tracking_prometheus_only: Optional[bool] = None +enable_end_user_cost_tracking_prometheus_only: Optional[bool] = None custom_prometheus_metadata_labels: List[str] = [] #### REQUEST PRIORITIZATION #### priority_reservation: Optional[Dict[str, float]] = None diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 1447fe1d81e..0b679619747 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -13274,133 +13274,5 @@ "max_output_tokens": 4096, "litellm_provider": "featherless_ai", "mode": "chat" - }, - "nebius/deepseek-ai/DeepSeek-V3-0324": { - "max_tokens": 128000, - "max_input_tokens": 64000, - "max_output_tokens": 64000, - "input_cost_per_token": 0.0000005, - "output_cost_per_token": 0.0000015, - "litellm_provider": "nebius", - "supports_function_calling": true, - "mode": "chat" - }, - "nebius/deepseek-ai/DeepSeek-V3-0324-fast": { - "max_tokens": 128000, - "max_input_tokens": 64000, - "max_output_tokens": 64000, - "input_cost_per_token": 0.000002, - "output_cost_per_token": 0.000006, - "litellm_provider": "nebius", - "supports_function_calling": true, - "mode": "chat" - }, - "nebius/deepseek-ai/DeepSeek-R1": { - "max_tokens": 128000, - "max_input_tokens": 64000, - "max_output_tokens": 64000, - "input_cost_per_token": 0.0000005, - "output_cost_per_token": 0.0000015, - "litellm_provider": "nebius", - "supports_function_calling": false, - "mode": "chat" - }, - "nebius/deepseek-ai/DeepSeek-R1-fast": { - "max_tokens": 128000, - "max_input_tokens": 64000, - "max_output_tokens": 64000, - "input_cost_per_token": 0.000002, - "output_cost_per_token": 0.000006, - "litellm_provider": "nebius", - "supports_function_calling": false, - "mode": "chat" - }, - "nebius/meta-llama/Llama-3.3-70B-Instruct": { - "max_tokens": 128000, - "max_input_tokens": 128000, - "max_output_tokens": 8192, - "input_cost_per_token": 0.00000013, - "output_cost_per_token": 0.0000004, - "litellm_provider": "nebius", - "supports_function_calling": true, - "mode": "chat" - }, - "nebius/Qwen/Qwen3-235B-A22B": { - "max_tokens": 40000, - "max_input_tokens": 32000, - "max_output_tokens": 8000, - "input_cost_per_token": 0.0000002, - "output_cost_per_token": 0.0000006, - "litellm_provider": "nebius", - "supports_function_calling": true, - "mode": "chat" - }, - "nebius/Qwen/Qwen3-30B-A3B": { - "max_tokens": 40000, - "max_input_tokens": 32000, - "max_output_tokens": 8000, - "input_cost_per_token": 0.0000001, - "output_cost_per_token": 0.0000003, - "litellm_provider": "nebius", - "supports_function_calling": true, - "mode": "chat" - }, - "nebius/Qwen/Qwen3-30B-A3B-fast": { - "max_tokens": 40000, - "max_input_tokens": 32000, - "max_output_tokens": 8000, - "input_cost_per_token": 0.0000003, - "output_cost_per_token": 0.0000009, - "litellm_provider": "nebius", - "supports_function_calling": true, - "mode": "chat" - }, - "nebius/Qwen/Qwen3-32B": { - "max_tokens": 40000, - "max_input_tokens": 32000, - "max_output_tokens": 8000, - "input_cost_per_token": 0.0000001, - "output_cost_per_token": 0.0000003, - "litellm_provider": "nebius", - "supports_function_calling": true, - "mode": "chat" - }, - "nebius/Qwen/Qwen2.5-VL-72B-Instruct": { - "max_tokens": 32000, - "max_input_tokens": 24000, - "max_output_tokens": 8000, - "input_cost_per_token": 0.00000025, - "output_cost_per_token": 0.00000075, - "litellm_provider": "nebius", - "supports_function_calling": true, - "supports_vision": true, - "mode": "chat" - }, - "nebius/google/gemma-3-27b-it-fast": { - "max_tokens": 128000, - "max_input_tokens": 64000, - "max_output_tokens": 64000, - "input_cost_per_token": 0.0000002, - "output_cost_per_token": 0.0000006, - "litellm_provider": "nebius", - "supports_function_calling": true, - "supports_vision": true, - "mode": "chat" - }, - "nebius/BAAI/bge-en-icl": { - "max_tokens": 32000, - "max_input_tokens": 32000, - "input_cost_per_token": 0.00000001, - "output_cost_per_token": 0.0, - "litellm_provider": "nebius", - "mode": "embedding" - }, - "nebius/BAAI/bge-multilingual-gemma2": { - "max_tokens": 32000, - "max_input_tokens": 32000, - "input_cost_per_token": 0.00000001, - "output_cost_per_token": 0.0, - "litellm_provider": "nebius", - "mode": "embedding" } } diff --git a/litellm/utils.py b/litellm/utils.py index ee09cd9bed3..fe9300aab51 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -6764,11 +6764,14 @@ def get_end_user_id_for_cost_tracking( ) if litellm.disable_end_user_cost_tracking: return None - if ( - service_type == "prometheus" - and litellm.disable_end_user_cost_tracking_prometheus_only - ): - return None + + ####################################### + # By default we don't track end_user on prometheus since we don't want to increase cardinality + # by default litellm.enable_end_user_cost_tracking_prometheus_only is None, so we don't track end_user on prometheus + ####################################### + if service_type == "prometheus": + if litellm.enable_end_user_cost_tracking_prometheus_only is not True: + return None return end_user_id diff --git a/tests/litellm_utils_tests/test_utils.py b/tests/litellm_utils_tests/test_utils.py index e6f45f5e928..c0a11bafe1e 100644 --- a/tests/litellm_utils_tests/test_utils.py +++ b/tests/litellm_utils_tests/test_utils.py @@ -1259,20 +1259,20 @@ def test_get_end_user_id_for_cost_tracking( @pytest.mark.parametrize( - "litellm_params, disable_end_user_cost_tracking_prometheus_only, expected_end_user_id", + "litellm_params, enable_end_user_cost_tracking_prometheus_only, expected_end_user_id", [ - ({}, False, None), - ({"user_api_key_end_user_id": "123"}, False, "123"), - ({"user_api_key_end_user_id": "123"}, True, None), + ({}, True, None), + ({"user_api_key_end_user_id": "123"}, True, "123"), + ({"user_api_key_end_user_id": "123"}, False, None), ], ) def test_get_end_user_id_for_cost_tracking_prometheus_only( - litellm_params, disable_end_user_cost_tracking_prometheus_only, expected_end_user_id + litellm_params, enable_end_user_cost_tracking_prometheus_only, expected_end_user_id ): from litellm.utils import get_end_user_id_for_cost_tracking - litellm.disable_end_user_cost_tracking_prometheus_only = ( - disable_end_user_cost_tracking_prometheus_only + litellm.enable_end_user_cost_tracking_prometheus_only = ( + enable_end_user_cost_tracking_prometheus_only ) assert ( get_end_user_id_for_cost_tracking( diff --git a/tests/logging_callback_tests/test_prometheus_unit_tests.py b/tests/logging_callback_tests/test_prometheus_unit_tests.py index 007de7e337e..ade57cc1017 100644 --- a/tests/logging_callback_tests/test_prometheus_unit_tests.py +++ b/tests/logging_callback_tests/test_prometheus_unit_tests.py @@ -706,7 +706,7 @@ async def test_async_post_call_failure_hook(prometheus_logger): # Assert failed requests metric was incremented with correct labels prometheus_logger.litellm_proxy_failed_requests_metric.labels.assert_called_once_with( - end_user="test_end_user", + end_user=None, hashed_api_key="test_key", api_key_alias="test_alias", requested_model="gpt-3.5-turbo", @@ -721,7 +721,7 @@ async def test_async_post_call_failure_hook(prometheus_logger): # Assert total requests metric was incremented with correct labels prometheus_logger.litellm_proxy_total_requests_metric.labels.assert_called_once_with( - end_user="test_end_user", + end_user=None, hashed_api_key="test_key", api_key_alias="test_alias", requested_model="gpt-3.5-turbo", @@ -767,7 +767,7 @@ async def test_async_post_call_success_hook(prometheus_logger): # Assert total requests metric was incremented with correct labels prometheus_logger.litellm_proxy_total_requests_metric.labels.assert_called_once_with( - end_user="test_end_user", + end_user=None, hashed_api_key="test_key", api_key_alias="test_alias", requested_model="gpt-3.5-turbo", @@ -1042,14 +1042,14 @@ def test_increment_deployment_cooled_down(prometheus_logger): prometheus_logger.litellm_deployment_cooled_down.labels().inc.assert_called_once() -@pytest.mark.parametrize("disable_end_user_tracking", [True, False]) -def test_prometheus_factory(monkeypatch, disable_end_user_tracking): +@pytest.mark.parametrize("enable_end_user_cost_tracking_prometheus_only", [True, False]) +def test_prometheus_factory(monkeypatch, enable_end_user_cost_tracking_prometheus_only): from litellm.integrations.prometheus import prometheus_label_factory from litellm.types.integrations.prometheus import UserAPIKeyLabelValues monkeypatch.setattr( - "litellm.disable_end_user_cost_tracking_prometheus_only", - disable_end_user_tracking, + "litellm.enable_end_user_cost_tracking_prometheus_only", + enable_end_user_cost_tracking_prometheus_only, ) enum_values = UserAPIKeyLabelValues( @@ -1062,10 +1062,10 @@ def test_prometheus_factory(monkeypatch, disable_end_user_tracking): supported_enum_labels=supported_labels, enum_values=enum_values ) - if disable_end_user_tracking: - assert returned_dict["end_user"] == None - else: + if enable_end_user_cost_tracking_prometheus_only is True: assert returned_dict["end_user"] == "test_end_user" + else: + assert returned_dict["end_user"] == None def test_get_custom_labels_from_metadata(monkeypatch): diff --git a/tests/test_litellm/integrations/test_prometheus.py b/tests/test_litellm/integrations/test_prometheus.py index 464477f0192..1a79e0a98d2 100644 --- a/tests/test_litellm/integrations/test_prometheus.py +++ b/tests/test_litellm/integrations/test_prometheus.py @@ -5,7 +5,6 @@ Mock prometheus unit tests, these don't rely on LLM API calls import json import os import sys - import pytest from fastapi.testclient import TestClient @@ -17,7 +16,11 @@ from apscheduler.schedulers.asyncio import AsyncIOScheduler import litellm from litellm.constants import PROMETHEUS_BUDGET_METRICS_REFRESH_INTERVAL_MINUTES -from litellm.integrations.prometheus import PrometheusLogger +from litellm.integrations.prometheus import PrometheusLogger, prometheus_label_factory +from litellm.types.integrations.prometheus import ( + PrometheusMetricLabels, + UserAPIKeyLabelValues, +) def test_initialize_budget_metrics_cron_job(): @@ -42,3 +45,140 @@ def test_initialize_budget_metrics_cron_job(): == PROMETHEUS_BUDGET_METRICS_REFRESH_INTERVAL_MINUTES ) assert job.func.__name__ == "initialize_remaining_budget_metrics" + + +def test_end_user_not_tracked_for_all_prometheus_metrics(): + """ + Test that end_user is not tracked for all Prometheus metrics by default. + + This test ensures that: + 1. By default, end_user is filtered out from all Prometheus metrics + 2. Future metrics that include end_user in their label definitions will also be filtered + 3. The filtering happens through the prometheus_label_factory function + """ + # Reset any previous settings + original_setting = getattr( + litellm, "enable_end_user_cost_tracking_prometheus_only", None + ) + litellm.enable_end_user_cost_tracking_prometheus_only = None # Default behavior + + try: + # Test data with end_user present + test_end_user_id = "test_user_123" + enum_values = UserAPIKeyLabelValues( + end_user=test_end_user_id, + hashed_api_key="test_key", + api_key_alias="test_alias", + team="test_team", + team_alias="test_team_alias", + user="test_user", + requested_model="gpt-4", + model="gpt-4", + litellm_model_name="gpt-4", + ) + + # Get all defined Prometheus metrics that include end_user in their labels + metrics_with_end_user = [] + for metric_name in PrometheusMetricLabels.__dict__: + if not metric_name.startswith("_") and metric_name != "get_labels": + labels = getattr(PrometheusMetricLabels, metric_name) + if isinstance(labels, list) and "end_user" in labels: + metrics_with_end_user.append(metric_name) + + # Ensure we found some metrics with end_user (sanity check) + assert ( + len(metrics_with_end_user) > 0 + ), "No metrics with end_user found - test setup issue" + + # Test each metric that includes end_user in its label definition + for metric_name in metrics_with_end_user: + supported_labels = PrometheusMetricLabels.get_labels(metric_name) + + # Verify that end_user is in the supported labels (before filtering) + assert ( + "end_user" in supported_labels + ), f"end_user should be in {metric_name} labels" + + # Call prometheus_label_factory to get filtered labels + filtered_labels = prometheus_label_factory( + supported_enum_labels=supported_labels, enum_values=enum_values + ) + print("filtered labels logged on prometheus=", filtered_labels) + + # Verify that end_user is None in the filtered labels (filtered out) + assert filtered_labels.get("end_user") is None, ( + f"end_user should be None for metric {metric_name} when " + f"enable_end_user_cost_tracking_prometheus_only is not True. " + f"Got: {filtered_labels.get('end_user')}" + ) + + # Test that when enable_end_user_cost_tracking_prometheus_only is True, end_user is tracked + litellm.enable_end_user_cost_tracking_prometheus_only = True + + # Test one metric to verify end_user is now included + test_metric = metrics_with_end_user[0] + supported_labels = PrometheusMetricLabels.get_labels(test_metric) + filtered_labels = prometheus_label_factory( + supported_enum_labels=supported_labels, enum_values=enum_values + ) + + # Now end_user should be present + assert filtered_labels.get("end_user") == test_end_user_id, ( + f"end_user should be present for metric {test_metric} when " + f"enable_end_user_cost_tracking_prometheus_only is True" + ) + + finally: + # Restore original setting + litellm.enable_end_user_cost_tracking_prometheus_only = original_setting + + +def test_future_metrics_with_end_user_are_filtered(): + """ + Test that ensures future metrics that include end_user will also be filtered. + This simulates adding a new metric with end_user in its labels. + """ + # Reset setting + original_setting = getattr( + litellm, "enable_end_user_cost_tracking_prometheus_only", None + ) + litellm.enable_end_user_cost_tracking_prometheus_only = None + + try: + # Simulate a new metric that includes end_user + simulated_new_metric_labels = [ + "end_user", + "hashed_api_key", + "api_key_alias", + "model", + "team", + "new_label", # Some new label that might be added in the future + ] + + test_end_user_id = "future_test_user" + enum_values = UserAPIKeyLabelValues( + end_user=test_end_user_id, + hashed_api_key="test_key", + api_key_alias="test_alias", + team="test_team", + model="gpt-4", + ) + + # Test the filtering + filtered_labels = prometheus_label_factory( + supported_enum_labels=simulated_new_metric_labels, enum_values=enum_values + ) + print("filtered labels logged on prometheus=", filtered_labels) + + # Verify end_user is filtered out even for this "new" metric + assert ( + filtered_labels.get("end_user") is None + ), "end_user should be filtered out for future metrics by default" + + # Verify other labels are present + assert filtered_labels.get("hashed_api_key") == "test_key" + assert filtered_labels.get("team") == "test_team" + + finally: + # Restore original setting + litellm.enable_end_user_cost_tracking_prometheus_only = original_setting