mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
[Fix] Prometheus Metrics - Do not track end_user by default + expose flag to enable tracking end_user on prometheus (#11192)
* fix: testing for disabling end user on metrics * fix: fixes for test_prometheus_factory * Delete litellm/model_prices_and_context_window_backup.json * fix: issues with merge conflicts * fix: test_get_end_user_id_for_cost_tracking_prometheus_only * Update tests/test_litellm/integrations/test_prometheus.py Co-authored-by: Copilot <175728472+Copilot@users.noreply.github.com> --------- Co-authored-by: Copilot <175728472+Copilot@users.noreply.github.com>
This commit is contained in:
parent
4c82dd9b27
commit
0590b1eb3a
6 changed files with 168 additions and 152 deletions
|
|
@ -296,6 +296,7 @@ tag_budget_config: Optional[Dict[str, BudgetConfig]] = None
|
|||
max_end_user_budget: Optional[float] = None
|
||||
disable_end_user_cost_tracking: Optional[bool] = None
|
||||
disable_end_user_cost_tracking_prometheus_only: Optional[bool] = None
|
||||
enable_end_user_cost_tracking_prometheus_only: Optional[bool] = None
|
||||
custom_prometheus_metadata_labels: List[str] = []
|
||||
#### REQUEST PRIORITIZATION ####
|
||||
priority_reservation: Optional[Dict[str, float]] = None
|
||||
|
|
|
|||
|
|
@ -13274,133 +13274,5 @@
|
|||
"max_output_tokens": 4096,
|
||||
"litellm_provider": "featherless_ai",
|
||||
"mode": "chat"
|
||||
},
|
||||
"nebius/deepseek-ai/DeepSeek-V3-0324": {
|
||||
"max_tokens": 128000,
|
||||
"max_input_tokens": 64000,
|
||||
"max_output_tokens": 64000,
|
||||
"input_cost_per_token": 0.0000005,
|
||||
"output_cost_per_token": 0.0000015,
|
||||
"litellm_provider": "nebius",
|
||||
"supports_function_calling": true,
|
||||
"mode": "chat"
|
||||
},
|
||||
"nebius/deepseek-ai/DeepSeek-V3-0324-fast": {
|
||||
"max_tokens": 128000,
|
||||
"max_input_tokens": 64000,
|
||||
"max_output_tokens": 64000,
|
||||
"input_cost_per_token": 0.000002,
|
||||
"output_cost_per_token": 0.000006,
|
||||
"litellm_provider": "nebius",
|
||||
"supports_function_calling": true,
|
||||
"mode": "chat"
|
||||
},
|
||||
"nebius/deepseek-ai/DeepSeek-R1": {
|
||||
"max_tokens": 128000,
|
||||
"max_input_tokens": 64000,
|
||||
"max_output_tokens": 64000,
|
||||
"input_cost_per_token": 0.0000005,
|
||||
"output_cost_per_token": 0.0000015,
|
||||
"litellm_provider": "nebius",
|
||||
"supports_function_calling": false,
|
||||
"mode": "chat"
|
||||
},
|
||||
"nebius/deepseek-ai/DeepSeek-R1-fast": {
|
||||
"max_tokens": 128000,
|
||||
"max_input_tokens": 64000,
|
||||
"max_output_tokens": 64000,
|
||||
"input_cost_per_token": 0.000002,
|
||||
"output_cost_per_token": 0.000006,
|
||||
"litellm_provider": "nebius",
|
||||
"supports_function_calling": false,
|
||||
"mode": "chat"
|
||||
},
|
||||
"nebius/meta-llama/Llama-3.3-70B-Instruct": {
|
||||
"max_tokens": 128000,
|
||||
"max_input_tokens": 128000,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 0.00000013,
|
||||
"output_cost_per_token": 0.0000004,
|
||||
"litellm_provider": "nebius",
|
||||
"supports_function_calling": true,
|
||||
"mode": "chat"
|
||||
},
|
||||
"nebius/Qwen/Qwen3-235B-A22B": {
|
||||
"max_tokens": 40000,
|
||||
"max_input_tokens": 32000,
|
||||
"max_output_tokens": 8000,
|
||||
"input_cost_per_token": 0.0000002,
|
||||
"output_cost_per_token": 0.0000006,
|
||||
"litellm_provider": "nebius",
|
||||
"supports_function_calling": true,
|
||||
"mode": "chat"
|
||||
},
|
||||
"nebius/Qwen/Qwen3-30B-A3B": {
|
||||
"max_tokens": 40000,
|
||||
"max_input_tokens": 32000,
|
||||
"max_output_tokens": 8000,
|
||||
"input_cost_per_token": 0.0000001,
|
||||
"output_cost_per_token": 0.0000003,
|
||||
"litellm_provider": "nebius",
|
||||
"supports_function_calling": true,
|
||||
"mode": "chat"
|
||||
},
|
||||
"nebius/Qwen/Qwen3-30B-A3B-fast": {
|
||||
"max_tokens": 40000,
|
||||
"max_input_tokens": 32000,
|
||||
"max_output_tokens": 8000,
|
||||
"input_cost_per_token": 0.0000003,
|
||||
"output_cost_per_token": 0.0000009,
|
||||
"litellm_provider": "nebius",
|
||||
"supports_function_calling": true,
|
||||
"mode": "chat"
|
||||
},
|
||||
"nebius/Qwen/Qwen3-32B": {
|
||||
"max_tokens": 40000,
|
||||
"max_input_tokens": 32000,
|
||||
"max_output_tokens": 8000,
|
||||
"input_cost_per_token": 0.0000001,
|
||||
"output_cost_per_token": 0.0000003,
|
||||
"litellm_provider": "nebius",
|
||||
"supports_function_calling": true,
|
||||
"mode": "chat"
|
||||
},
|
||||
"nebius/Qwen/Qwen2.5-VL-72B-Instruct": {
|
||||
"max_tokens": 32000,
|
||||
"max_input_tokens": 24000,
|
||||
"max_output_tokens": 8000,
|
||||
"input_cost_per_token": 0.00000025,
|
||||
"output_cost_per_token": 0.00000075,
|
||||
"litellm_provider": "nebius",
|
||||
"supports_function_calling": true,
|
||||
"supports_vision": true,
|
||||
"mode": "chat"
|
||||
},
|
||||
"nebius/google/gemma-3-27b-it-fast": {
|
||||
"max_tokens": 128000,
|
||||
"max_input_tokens": 64000,
|
||||
"max_output_tokens": 64000,
|
||||
"input_cost_per_token": 0.0000002,
|
||||
"output_cost_per_token": 0.0000006,
|
||||
"litellm_provider": "nebius",
|
||||
"supports_function_calling": true,
|
||||
"supports_vision": true,
|
||||
"mode": "chat"
|
||||
},
|
||||
"nebius/BAAI/bge-en-icl": {
|
||||
"max_tokens": 32000,
|
||||
"max_input_tokens": 32000,
|
||||
"input_cost_per_token": 0.00000001,
|
||||
"output_cost_per_token": 0.0,
|
||||
"litellm_provider": "nebius",
|
||||
"mode": "embedding"
|
||||
},
|
||||
"nebius/BAAI/bge-multilingual-gemma2": {
|
||||
"max_tokens": 32000,
|
||||
"max_input_tokens": 32000,
|
||||
"input_cost_per_token": 0.00000001,
|
||||
"output_cost_per_token": 0.0,
|
||||
"litellm_provider": "nebius",
|
||||
"mode": "embedding"
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -6764,11 +6764,14 @@ def get_end_user_id_for_cost_tracking(
|
|||
)
|
||||
if litellm.disable_end_user_cost_tracking:
|
||||
return None
|
||||
if (
|
||||
service_type == "prometheus"
|
||||
and litellm.disable_end_user_cost_tracking_prometheus_only
|
||||
):
|
||||
return None
|
||||
|
||||
#######################################
|
||||
# By default we don't track end_user on prometheus since we don't want to increase cardinality
|
||||
# by default litellm.enable_end_user_cost_tracking_prometheus_only is None, so we don't track end_user on prometheus
|
||||
#######################################
|
||||
if service_type == "prometheus":
|
||||
if litellm.enable_end_user_cost_tracking_prometheus_only is not True:
|
||||
return None
|
||||
return end_user_id
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -1259,20 +1259,20 @@ def test_get_end_user_id_for_cost_tracking(
|
|||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"litellm_params, disable_end_user_cost_tracking_prometheus_only, expected_end_user_id",
|
||||
"litellm_params, enable_end_user_cost_tracking_prometheus_only, expected_end_user_id",
|
||||
[
|
||||
({}, False, None),
|
||||
({"user_api_key_end_user_id": "123"}, False, "123"),
|
||||
({"user_api_key_end_user_id": "123"}, True, None),
|
||||
({}, True, None),
|
||||
({"user_api_key_end_user_id": "123"}, True, "123"),
|
||||
({"user_api_key_end_user_id": "123"}, False, None),
|
||||
],
|
||||
)
|
||||
def test_get_end_user_id_for_cost_tracking_prometheus_only(
|
||||
litellm_params, disable_end_user_cost_tracking_prometheus_only, expected_end_user_id
|
||||
litellm_params, enable_end_user_cost_tracking_prometheus_only, expected_end_user_id
|
||||
):
|
||||
from litellm.utils import get_end_user_id_for_cost_tracking
|
||||
|
||||
litellm.disable_end_user_cost_tracking_prometheus_only = (
|
||||
disable_end_user_cost_tracking_prometheus_only
|
||||
litellm.enable_end_user_cost_tracking_prometheus_only = (
|
||||
enable_end_user_cost_tracking_prometheus_only
|
||||
)
|
||||
assert (
|
||||
get_end_user_id_for_cost_tracking(
|
||||
|
|
|
|||
|
|
@ -706,7 +706,7 @@ async def test_async_post_call_failure_hook(prometheus_logger):
|
|||
|
||||
# Assert failed requests metric was incremented with correct labels
|
||||
prometheus_logger.litellm_proxy_failed_requests_metric.labels.assert_called_once_with(
|
||||
end_user="test_end_user",
|
||||
end_user=None,
|
||||
hashed_api_key="test_key",
|
||||
api_key_alias="test_alias",
|
||||
requested_model="gpt-3.5-turbo",
|
||||
|
|
@ -721,7 +721,7 @@ async def test_async_post_call_failure_hook(prometheus_logger):
|
|||
|
||||
# Assert total requests metric was incremented with correct labels
|
||||
prometheus_logger.litellm_proxy_total_requests_metric.labels.assert_called_once_with(
|
||||
end_user="test_end_user",
|
||||
end_user=None,
|
||||
hashed_api_key="test_key",
|
||||
api_key_alias="test_alias",
|
||||
requested_model="gpt-3.5-turbo",
|
||||
|
|
@ -767,7 +767,7 @@ async def test_async_post_call_success_hook(prometheus_logger):
|
|||
|
||||
# Assert total requests metric was incremented with correct labels
|
||||
prometheus_logger.litellm_proxy_total_requests_metric.labels.assert_called_once_with(
|
||||
end_user="test_end_user",
|
||||
end_user=None,
|
||||
hashed_api_key="test_key",
|
||||
api_key_alias="test_alias",
|
||||
requested_model="gpt-3.5-turbo",
|
||||
|
|
@ -1042,14 +1042,14 @@ def test_increment_deployment_cooled_down(prometheus_logger):
|
|||
prometheus_logger.litellm_deployment_cooled_down.labels().inc.assert_called_once()
|
||||
|
||||
|
||||
@pytest.mark.parametrize("disable_end_user_tracking", [True, False])
|
||||
def test_prometheus_factory(monkeypatch, disable_end_user_tracking):
|
||||
@pytest.mark.parametrize("enable_end_user_cost_tracking_prometheus_only", [True, False])
|
||||
def test_prometheus_factory(monkeypatch, enable_end_user_cost_tracking_prometheus_only):
|
||||
from litellm.integrations.prometheus import prometheus_label_factory
|
||||
from litellm.types.integrations.prometheus import UserAPIKeyLabelValues
|
||||
|
||||
monkeypatch.setattr(
|
||||
"litellm.disable_end_user_cost_tracking_prometheus_only",
|
||||
disable_end_user_tracking,
|
||||
"litellm.enable_end_user_cost_tracking_prometheus_only",
|
||||
enable_end_user_cost_tracking_prometheus_only,
|
||||
)
|
||||
|
||||
enum_values = UserAPIKeyLabelValues(
|
||||
|
|
@ -1062,10 +1062,10 @@ def test_prometheus_factory(monkeypatch, disable_end_user_tracking):
|
|||
supported_enum_labels=supported_labels, enum_values=enum_values
|
||||
)
|
||||
|
||||
if disable_end_user_tracking:
|
||||
assert returned_dict["end_user"] == None
|
||||
else:
|
||||
if enable_end_user_cost_tracking_prometheus_only is True:
|
||||
assert returned_dict["end_user"] == "test_end_user"
|
||||
else:
|
||||
assert returned_dict["end_user"] == None
|
||||
|
||||
|
||||
def test_get_custom_labels_from_metadata(monkeypatch):
|
||||
|
|
|
|||
|
|
@ -5,7 +5,6 @@ Mock prometheus unit tests, these don't rely on LLM API calls
|
|||
import json
|
||||
import os
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
|
|
@ -17,7 +16,11 @@ from apscheduler.schedulers.asyncio import AsyncIOScheduler
|
|||
|
||||
import litellm
|
||||
from litellm.constants import PROMETHEUS_BUDGET_METRICS_REFRESH_INTERVAL_MINUTES
|
||||
from litellm.integrations.prometheus import PrometheusLogger
|
||||
from litellm.integrations.prometheus import PrometheusLogger, prometheus_label_factory
|
||||
from litellm.types.integrations.prometheus import (
|
||||
PrometheusMetricLabels,
|
||||
UserAPIKeyLabelValues,
|
||||
)
|
||||
|
||||
|
||||
def test_initialize_budget_metrics_cron_job():
|
||||
|
|
@ -42,3 +45,140 @@ def test_initialize_budget_metrics_cron_job():
|
|||
== PROMETHEUS_BUDGET_METRICS_REFRESH_INTERVAL_MINUTES
|
||||
)
|
||||
assert job.func.__name__ == "initialize_remaining_budget_metrics"
|
||||
|
||||
|
||||
def test_end_user_not_tracked_for_all_prometheus_metrics():
|
||||
"""
|
||||
Test that end_user is not tracked for all Prometheus metrics by default.
|
||||
|
||||
This test ensures that:
|
||||
1. By default, end_user is filtered out from all Prometheus metrics
|
||||
2. Future metrics that include end_user in their label definitions will also be filtered
|
||||
3. The filtering happens through the prometheus_label_factory function
|
||||
"""
|
||||
# Reset any previous settings
|
||||
original_setting = getattr(
|
||||
litellm, "enable_end_user_cost_tracking_prometheus_only", None
|
||||
)
|
||||
litellm.enable_end_user_cost_tracking_prometheus_only = None # Default behavior
|
||||
|
||||
try:
|
||||
# Test data with end_user present
|
||||
test_end_user_id = "test_user_123"
|
||||
enum_values = UserAPIKeyLabelValues(
|
||||
end_user=test_end_user_id,
|
||||
hashed_api_key="test_key",
|
||||
api_key_alias="test_alias",
|
||||
team="test_team",
|
||||
team_alias="test_team_alias",
|
||||
user="test_user",
|
||||
requested_model="gpt-4",
|
||||
model="gpt-4",
|
||||
litellm_model_name="gpt-4",
|
||||
)
|
||||
|
||||
# Get all defined Prometheus metrics that include end_user in their labels
|
||||
metrics_with_end_user = []
|
||||
for metric_name in PrometheusMetricLabels.__dict__:
|
||||
if not metric_name.startswith("_") and metric_name != "get_labels":
|
||||
labels = getattr(PrometheusMetricLabels, metric_name)
|
||||
if isinstance(labels, list) and "end_user" in labels:
|
||||
metrics_with_end_user.append(metric_name)
|
||||
|
||||
# Ensure we found some metrics with end_user (sanity check)
|
||||
assert (
|
||||
len(metrics_with_end_user) > 0
|
||||
), "No metrics with end_user found - test setup issue"
|
||||
|
||||
# Test each metric that includes end_user in its label definition
|
||||
for metric_name in metrics_with_end_user:
|
||||
supported_labels = PrometheusMetricLabels.get_labels(metric_name)
|
||||
|
||||
# Verify that end_user is in the supported labels (before filtering)
|
||||
assert (
|
||||
"end_user" in supported_labels
|
||||
), f"end_user should be in {metric_name} labels"
|
||||
|
||||
# Call prometheus_label_factory to get filtered labels
|
||||
filtered_labels = prometheus_label_factory(
|
||||
supported_enum_labels=supported_labels, enum_values=enum_values
|
||||
)
|
||||
print("filtered labels logged on prometheus=", filtered_labels)
|
||||
|
||||
# Verify that end_user is None in the filtered labels (filtered out)
|
||||
assert filtered_labels.get("end_user") is None, (
|
||||
f"end_user should be None for metric {metric_name} when "
|
||||
f"enable_end_user_cost_tracking_prometheus_only is not True. "
|
||||
f"Got: {filtered_labels.get('end_user')}"
|
||||
)
|
||||
|
||||
# Test that when enable_end_user_cost_tracking_prometheus_only is True, end_user is tracked
|
||||
litellm.enable_end_user_cost_tracking_prometheus_only = True
|
||||
|
||||
# Test one metric to verify end_user is now included
|
||||
test_metric = metrics_with_end_user[0]
|
||||
supported_labels = PrometheusMetricLabels.get_labels(test_metric)
|
||||
filtered_labels = prometheus_label_factory(
|
||||
supported_enum_labels=supported_labels, enum_values=enum_values
|
||||
)
|
||||
|
||||
# Now end_user should be present
|
||||
assert filtered_labels.get("end_user") == test_end_user_id, (
|
||||
f"end_user should be present for metric {test_metric} when "
|
||||
f"enable_end_user_cost_tracking_prometheus_only is True"
|
||||
)
|
||||
|
||||
finally:
|
||||
# Restore original setting
|
||||
litellm.enable_end_user_cost_tracking_prometheus_only = original_setting
|
||||
|
||||
|
||||
def test_future_metrics_with_end_user_are_filtered():
|
||||
"""
|
||||
Test that ensures future metrics that include end_user will also be filtered.
|
||||
This simulates adding a new metric with end_user in its labels.
|
||||
"""
|
||||
# Reset setting
|
||||
original_setting = getattr(
|
||||
litellm, "enable_end_user_cost_tracking_prometheus_only", None
|
||||
)
|
||||
litellm.enable_end_user_cost_tracking_prometheus_only = None
|
||||
|
||||
try:
|
||||
# Simulate a new metric that includes end_user
|
||||
simulated_new_metric_labels = [
|
||||
"end_user",
|
||||
"hashed_api_key",
|
||||
"api_key_alias",
|
||||
"model",
|
||||
"team",
|
||||
"new_label", # Some new label that might be added in the future
|
||||
]
|
||||
|
||||
test_end_user_id = "future_test_user"
|
||||
enum_values = UserAPIKeyLabelValues(
|
||||
end_user=test_end_user_id,
|
||||
hashed_api_key="test_key",
|
||||
api_key_alias="test_alias",
|
||||
team="test_team",
|
||||
model="gpt-4",
|
||||
)
|
||||
|
||||
# Test the filtering
|
||||
filtered_labels = prometheus_label_factory(
|
||||
supported_enum_labels=simulated_new_metric_labels, enum_values=enum_values
|
||||
)
|
||||
print("filtered labels logged on prometheus=", filtered_labels)
|
||||
|
||||
# Verify end_user is filtered out even for this "new" metric
|
||||
assert (
|
||||
filtered_labels.get("end_user") is None
|
||||
), "end_user should be filtered out for future metrics by default"
|
||||
|
||||
# Verify other labels are present
|
||||
assert filtered_labels.get("hashed_api_key") == "test_key"
|
||||
assert filtered_labels.get("team") == "test_team"
|
||||
|
||||
finally:
|
||||
# Restore original setting
|
||||
litellm.enable_end_user_cost_tracking_prometheus_only = original_setting
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue