[Fix] Prometheus Metrics - Do not track end_user by default + expose flag to enable tracking end_user on prometheus (#11192)

* fix: testing for disabling end user on metrics

* fix: fixes for test_prometheus_factory

* Delete litellm/model_prices_and_context_window_backup.json

* fix: issues with merge conflicts

* fix: test_get_end_user_id_for_cost_tracking_prometheus_only

* Update tests/test_litellm/integrations/test_prometheus.py

Co-authored-by: Copilot <175728472+Copilot@users.noreply.github.com>

---------

Co-authored-by: Copilot <175728472+Copilot@users.noreply.github.com>
This commit is contained in:
Ishaan Jaff 2025-05-27 17:06:58 -07:00 • committed by GitHub
parent 4c82dd9b27
commit 0590b1eb3a
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
6 changed files with 168 additions and 152 deletions

View file

@ -296,6 +296,7 @@ tag_budget_config: Optional[Dict[str, BudgetConfig]] = None
max_end_user_budget: Optional[float] = None
disable_end_user_cost_tracking: Optional[bool] = None
disable_end_user_cost_tracking_prometheus_only: Optional[bool] = None
enable_end_user_cost_tracking_prometheus_only: Optional[bool] = None
custom_prometheus_metadata_labels: List[str] = []
#### REQUEST PRIORITIZATION ####
priority_reservation: Optional[Dict[str, float]] = None

View file

@ -13274,133 +13274,5 @@
"max_output_tokens": 4096,
"litellm_provider": "featherless_ai",
"mode": "chat"
},
"nebius/deepseek-ai/DeepSeek-V3-0324": {
"max_tokens": 128000,
"max_input_tokens": 64000,
"max_output_tokens": 64000,
"input_cost_per_token": 0.0000005,
"output_cost_per_token": 0.0000015,
"litellm_provider": "nebius",
"supports_function_calling": true,
"mode": "chat"
},
"nebius/deepseek-ai/DeepSeek-V3-0324-fast": {
"max_tokens": 128000,
"max_input_tokens": 64000,
"max_output_tokens": 64000,
"input_cost_per_token": 0.000002,
"output_cost_per_token": 0.000006,
"litellm_provider": "nebius",
"supports_function_calling": true,
"mode": "chat"
},
"nebius/deepseek-ai/DeepSeek-R1": {
"max_tokens": 128000,
"max_input_tokens": 64000,
"max_output_tokens": 64000,
"input_cost_per_token": 0.0000005,
"output_cost_per_token": 0.0000015,
"litellm_provider": "nebius",
"supports_function_calling": false,
"mode": "chat"
},
"nebius/deepseek-ai/DeepSeek-R1-fast": {
"max_tokens": 128000,
"max_input_tokens": 64000,
"max_output_tokens": 64000,
"input_cost_per_token": 0.000002,
"output_cost_per_token": 0.000006,
"litellm_provider": "nebius",
"supports_function_calling": false,
"mode": "chat"
},
"nebius/meta-llama/Llama-3.3-70B-Instruct": {
"max_tokens": 128000,
"max_input_tokens": 128000,
"max_output_tokens": 8192,
"input_cost_per_token": 0.00000013,
"output_cost_per_token": 0.0000004,
"litellm_provider": "nebius",
"supports_function_calling": true,
"mode": "chat"
},
"nebius/Qwen/Qwen3-235B-A22B": {
"max_tokens": 40000,
"max_input_tokens": 32000,
"max_output_tokens": 8000,
"input_cost_per_token": 0.0000002,
"output_cost_per_token": 0.0000006,
"litellm_provider": "nebius",
"supports_function_calling": true,
"mode": "chat"
},
"nebius/Qwen/Qwen3-30B-A3B": {
"max_tokens": 40000,
"max_input_tokens": 32000,
"max_output_tokens": 8000,
"input_cost_per_token": 0.0000001,
"output_cost_per_token": 0.0000003,
"litellm_provider": "nebius",
"supports_function_calling": true,
"mode": "chat"
},
"nebius/Qwen/Qwen3-30B-A3B-fast": {
"max_tokens": 40000,
"max_input_tokens": 32000,
"max_output_tokens": 8000,
"input_cost_per_token": 0.0000003,
"output_cost_per_token": 0.0000009,
"litellm_provider": "nebius",
"supports_function_calling": true,
"mode": "chat"
},
"nebius/Qwen/Qwen3-32B": {
"max_tokens": 40000,
"max_input_tokens": 32000,
"max_output_tokens": 8000,
"input_cost_per_token": 0.0000001,
"output_cost_per_token": 0.0000003,
"litellm_provider": "nebius",
"supports_function_calling": true,
"mode": "chat"
},
"nebius/Qwen/Qwen2.5-VL-72B-Instruct": {
"max_tokens": 32000,
"max_input_tokens": 24000,
"max_output_tokens": 8000,
"input_cost_per_token": 0.00000025,
"output_cost_per_token": 0.00000075,
"litellm_provider": "nebius",
"supports_function_calling": true,
"supports_vision": true,
"mode": "chat"
},
"nebius/google/gemma-3-27b-it-fast": {
"max_tokens": 128000,
"max_input_tokens": 64000,
"max_output_tokens": 64000,
"input_cost_per_token": 0.0000002,
"output_cost_per_token": 0.0000006,
"litellm_provider": "nebius",
"supports_function_calling": true,
"supports_vision": true,
"mode": "chat"
},
"nebius/BAAI/bge-en-icl": {
"max_tokens": 32000,
"max_input_tokens": 32000,
"input_cost_per_token": 0.00000001,
"output_cost_per_token": 0.0,
"litellm_provider": "nebius",
"mode": "embedding"
},
"nebius/BAAI/bge-multilingual-gemma2": {
"max_tokens": 32000,
"max_input_tokens": 32000,
"input_cost_per_token": 0.00000001,
"output_cost_per_token": 0.0,
"litellm_provider": "nebius",
"mode": "embedding"
}
}

View file

@ -6764,11 +6764,14 @@ def get_end_user_id_for_cost_tracking(
)
if litellm.disable_end_user_cost_tracking:
return None
if (
service_type == "prometheus"
and litellm.disable_end_user_cost_tracking_prometheus_only
):
return None
#######################################
# By default we don't track end_user on prometheus since we don't want to increase cardinality
# by default litellm.enable_end_user_cost_tracking_prometheus_only is None, so we don't track end_user on prometheus
#######################################
if service_type == "prometheus":
if litellm.enable_end_user_cost_tracking_prometheus_only is not True:
return None
return end_user_id

View file

@ -1259,20 +1259,20 @@ def test_get_end_user_id_for_cost_tracking(
@pytest.mark.parametrize(
"litellm_params, disable_end_user_cost_tracking_prometheus_only, expected_end_user_id",
"litellm_params, enable_end_user_cost_tracking_prometheus_only, expected_end_user_id",
[
({}, False, None),
({"user_api_key_end_user_id": "123"}, False, "123"),
({"user_api_key_end_user_id": "123"}, True, None),
({}, True, None),
({"user_api_key_end_user_id": "123"}, True, "123"),
({"user_api_key_end_user_id": "123"}, False, None),
],
)
def test_get_end_user_id_for_cost_tracking_prometheus_only(
litellm_params, disable_end_user_cost_tracking_prometheus_only, expected_end_user_id
litellm_params, enable_end_user_cost_tracking_prometheus_only, expected_end_user_id
):
from litellm.utils import get_end_user_id_for_cost_tracking
litellm.disable_end_user_cost_tracking_prometheus_only = (
disable_end_user_cost_tracking_prometheus_only
litellm.enable_end_user_cost_tracking_prometheus_only = (
enable_end_user_cost_tracking_prometheus_only
)
assert (
get_end_user_id_for_cost_tracking(

View file

@ -706,7 +706,7 @@ async def test_async_post_call_failure_hook(prometheus_logger):
# Assert failed requests metric was incremented with correct labels
prometheus_logger.litellm_proxy_failed_requests_metric.labels.assert_called_once_with(
end_user="test_end_user",
end_user=None,
hashed_api_key="test_key",
api_key_alias="test_alias",
requested_model="gpt-3.5-turbo",
@ -721,7 +721,7 @@ async def test_async_post_call_failure_hook(prometheus_logger):
# Assert total requests metric was incremented with correct labels
prometheus_logger.litellm_proxy_total_requests_metric.labels.assert_called_once_with(
end_user="test_end_user",
end_user=None,
hashed_api_key="test_key",
api_key_alias="test_alias",
requested_model="gpt-3.5-turbo",
@ -767,7 +767,7 @@ async def test_async_post_call_success_hook(prometheus_logger):
# Assert total requests metric was incremented with correct labels
prometheus_logger.litellm_proxy_total_requests_metric.labels.assert_called_once_with(
end_user="test_end_user",
end_user=None,
hashed_api_key="test_key",
api_key_alias="test_alias",
requested_model="gpt-3.5-turbo",
@ -1042,14 +1042,14 @@ def test_increment_deployment_cooled_down(prometheus_logger):
prometheus_logger.litellm_deployment_cooled_down.labels().inc.assert_called_once()
@pytest.mark.parametrize("disable_end_user_tracking", [True, False])
def test_prometheus_factory(monkeypatch, disable_end_user_tracking):
@pytest.mark.parametrize("enable_end_user_cost_tracking_prometheus_only", [True, False])
def test_prometheus_factory(monkeypatch, enable_end_user_cost_tracking_prometheus_only):
from litellm.integrations.prometheus import prometheus_label_factory
from litellm.types.integrations.prometheus import UserAPIKeyLabelValues
monkeypatch.setattr(
"litellm.disable_end_user_cost_tracking_prometheus_only",
disable_end_user_tracking,
"litellm.enable_end_user_cost_tracking_prometheus_only",
enable_end_user_cost_tracking_prometheus_only,
)
enum_values = UserAPIKeyLabelValues(
@ -1062,10 +1062,10 @@ def test_prometheus_factory(monkeypatch, disable_end_user_tracking):
supported_enum_labels=supported_labels, enum_values=enum_values
)
if disable_end_user_tracking:
assert returned_dict["end_user"] == None
else:
if enable_end_user_cost_tracking_prometheus_only is True:
assert returned_dict["end_user"] == "test_end_user"
else:
assert returned_dict["end_user"] == None
def test_get_custom_labels_from_metadata(monkeypatch):

View file

@ -5,7 +5,6 @@ Mock prometheus unit tests, these don't rely on LLM API calls
import json
import os
import sys
import pytest
from fastapi.testclient import TestClient
@ -17,7 +16,11 @@ from apscheduler.schedulers.asyncio import AsyncIOScheduler
import litellm
from litellm.constants import PROMETHEUS_BUDGET_METRICS_REFRESH_INTERVAL_MINUTES
from litellm.integrations.prometheus import PrometheusLogger
from litellm.integrations.prometheus import PrometheusLogger, prometheus_label_factory
from litellm.types.integrations.prometheus import (
PrometheusMetricLabels,
UserAPIKeyLabelValues,
)
def test_initialize_budget_metrics_cron_job():
@ -42,3 +45,140 @@ def test_initialize_budget_metrics_cron_job():
== PROMETHEUS_BUDGET_METRICS_REFRESH_INTERVAL_MINUTES
)
assert job.func.__name__ == "initialize_remaining_budget_metrics"
def test_end_user_not_tracked_for_all_prometheus_metrics():
"""
Test that end_user is not tracked for all Prometheus metrics by default.
This test ensures that:
1. By default, end_user is filtered out from all Prometheus metrics
2. Future metrics that include end_user in their label definitions will also be filtered
3. The filtering happens through the prometheus_label_factory function
"""
# Reset any previous settings
original_setting = getattr(
litellm, "enable_end_user_cost_tracking_prometheus_only", None
)
litellm.enable_end_user_cost_tracking_prometheus_only = None # Default behavior
try:
# Test data with end_user present
test_end_user_id = "test_user_123"
enum_values = UserAPIKeyLabelValues(
end_user=test_end_user_id,
hashed_api_key="test_key",
api_key_alias="test_alias",
team="test_team",
team_alias="test_team_alias",
user="test_user",
requested_model="gpt-4",
model="gpt-4",
litellm_model_name="gpt-4",
)
# Get all defined Prometheus metrics that include end_user in their labels
metrics_with_end_user = []
for metric_name in PrometheusMetricLabels.__dict__:
if not metric_name.startswith("_") and metric_name != "get_labels":
labels = getattr(PrometheusMetricLabels, metric_name)
if isinstance(labels, list) and "end_user" in labels:
metrics_with_end_user.append(metric_name)
# Ensure we found some metrics with end_user (sanity check)
assert (
len(metrics_with_end_user) > 0
), "No metrics with end_user found - test setup issue"
# Test each metric that includes end_user in its label definition
for metric_name in metrics_with_end_user:
supported_labels = PrometheusMetricLabels.get_labels(metric_name)
# Verify that end_user is in the supported labels (before filtering)
assert (
"end_user" in supported_labels
), f"end_user should be in {metric_name} labels"
# Call prometheus_label_factory to get filtered labels
filtered_labels = prometheus_label_factory(
supported_enum_labels=supported_labels, enum_values=enum_values
)
print("filtered labels logged on prometheus=", filtered_labels)
# Verify that end_user is None in the filtered labels (filtered out)
assert filtered_labels.get("end_user") is None, (
f"end_user should be None for metric {metric_name} when "
f"enable_end_user_cost_tracking_prometheus_only is not True. "
f"Got: {filtered_labels.get('end_user')}"
)
# Test that when enable_end_user_cost_tracking_prometheus_only is True, end_user is tracked
litellm.enable_end_user_cost_tracking_prometheus_only = True
# Test one metric to verify end_user is now included
test_metric = metrics_with_end_user[0]
supported_labels = PrometheusMetricLabels.get_labels(test_metric)
filtered_labels = prometheus_label_factory(
supported_enum_labels=supported_labels, enum_values=enum_values
)
# Now end_user should be present
assert filtered_labels.get("end_user") == test_end_user_id, (
f"end_user should be present for metric {test_metric} when "
f"enable_end_user_cost_tracking_prometheus_only is True"
)
finally:
# Restore original setting
litellm.enable_end_user_cost_tracking_prometheus_only = original_setting
def test_future_metrics_with_end_user_are_filtered():
"""
Test that ensures future metrics that include end_user will also be filtered.
This simulates adding a new metric with end_user in its labels.
"""
# Reset setting
original_setting = getattr(
litellm, "enable_end_user_cost_tracking_prometheus_only", None
)
litellm.enable_end_user_cost_tracking_prometheus_only = None
try:
# Simulate a new metric that includes end_user
simulated_new_metric_labels = [
"end_user",
"hashed_api_key",
"api_key_alias",
"model",
"team",
"new_label", # Some new label that might be added in the future
]
test_end_user_id = "future_test_user"
enum_values = UserAPIKeyLabelValues(
end_user=test_end_user_id,
hashed_api_key="test_key",
api_key_alias="test_alias",
team="test_team",
model="gpt-4",
)
# Test the filtering
filtered_labels = prometheus_label_factory(
supported_enum_labels=simulated_new_metric_labels, enum_values=enum_values
)
print("filtered labels logged on prometheus=", filtered_labels)
# Verify end_user is filtered out even for this "new" metric
assert (
filtered_labels.get("end_user") is None
), "end_user should be filtered out for future metrics by default"
# Verify other labels are present
assert filtered_labels.get("hashed_api_key") == "test_key"
assert filtered_labels.get("team") == "test_team"
finally:
# Restore original setting
litellm.enable_end_user_cost_tracking_prometheus_only = original_setting