From 5f80e8d5e8b52a647ec757b630c134c4b097ed2b Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Tue, 20 Jan 2026 15:28:09 +0530 Subject: [PATCH] Fix for Prometheus Metric Cardinality Issue with /responses Endpoint --- litellm/proxy/auth/auth_utils.py | 82 +++++++++ litellm/proxy/auth/user_api_key_auth.py | 5 +- .../integrations/test_prometheus_labels.py | 160 +++++++++++++++++- 3 files changed, 242 insertions(+), 5 deletions(-) diff --git a/litellm/proxy/auth/auth_utils.py b/litellm/proxy/auth/auth_utils.py index 1a7f05716b3..9b9a988c07a 100644 --- a/litellm/proxy/auth/auth_utils.py +++ b/litellm/proxy/auth/auth_utils.py @@ -311,6 +311,88 @@ def get_request_route(request: Request) -> str: return request.url.path +def normalize_request_route(route: str) -> str: + """ + Normalize request routes by replacing dynamic path parameters with placeholders. + + This prevents high cardinality in Prometheus metrics by collapsing routes like: + - /v1/responses/1234567890 -> /v1/responses/{response_id} + - /v1/threads/thread_123 -> /v1/threads/{thread_id} + + Args: + route: The request route path + + Returns: + Normalized route with dynamic parameters replaced by placeholders + + Examples: + >>> normalize_request_route("/v1/responses/abc123") + '/v1/responses/{response_id}' + >>> normalize_request_route("/v1/responses/abc123/cancel") + '/v1/responses/{response_id}/cancel' + >>> normalize_request_route("/chat/completions") + '/chat/completions' + """ + # Define patterns for routes with dynamic IDs + # Format: (regex_pattern, replacement_template) + patterns = [ + # Responses API - must come before generic patterns + (r'^(/(?:openai/)?v1/responses)/([^/]+)(/input_items)$', r'\1/{response_id}\3'), + (r'^(/(?:openai/)?v1/responses)/([^/]+)(/cancel)$', r'\1/{response_id}\3'), + (r'^(/(?:openai/)?v1/responses)/([^/]+)$', r'\1/{response_id}'), + (r'^(/responses)/([^/]+)(/input_items)$', r'\1/{response_id}\3'), + (r'^(/responses)/([^/]+)(/cancel)$', r'\1/{response_id}\3'), + (r'^(/responses)/([^/]+)$', r'\1/{response_id}'), + + # Threads API + (r'^(/(?:openai/)?v1/threads)/([^/]+)(/runs)/([^/]+)(/steps)/([^/]+)$', r'\1/{thread_id}\3/{run_id}\5/{step_id}'), + (r'^(/(?:openai/)?v1/threads)/([^/]+)(/runs)/([^/]+)(/steps)$', r'\1/{thread_id}\3/{run_id}\5'), + (r'^(/(?:openai/)?v1/threads)/([^/]+)(/runs)/([^/]+)(/cancel)$', r'\1/{thread_id}\3/{run_id}\5'), + (r'^(/(?:openai/)?v1/threads)/([^/]+)(/runs)/([^/]+)(/submit_tool_outputs)$', r'\1/{thread_id}\3/{run_id}\5'), + (r'^(/(?:openai/)?v1/threads)/([^/]+)(/runs)/([^/]+)$', r'\1/{thread_id}\3/{run_id}'), + (r'^(/(?:openai/)?v1/threads)/([^/]+)(/runs)$', r'\1/{thread_id}\3'), + (r'^(/(?:openai/)?v1/threads)/([^/]+)(/messages)/([^/]+)$', r'\1/{thread_id}\3/{message_id}'), + (r'^(/(?:openai/)?v1/threads)/([^/]+)(/messages)$', r'\1/{thread_id}\3'), + (r'^(/(?:openai/)?v1/threads)/([^/]+)$', r'\1/{thread_id}'), + + # Vector Stores API + (r'^(/(?:openai/)?v1/vector_stores)/([^/]+)(/files)/([^/]+)$', r'\1/{vector_store_id}\3/{file_id}'), + (r'^(/(?:openai/)?v1/vector_stores)/([^/]+)(/files)$', r'\1/{vector_store_id}\3'), + (r'^(/(?:openai/)?v1/vector_stores)/([^/]+)(/file_batches)/([^/]+)$', r'\1/{vector_store_id}\3/{batch_id}'), + (r'^(/(?:openai/)?v1/vector_stores)/([^/]+)(/file_batches)$', r'\1/{vector_store_id}\3'), + (r'^(/(?:openai/)?v1/vector_stores)/([^/]+)$', r'\1/{vector_store_id}'), + + # Assistants API + (r'^(/(?:openai/)?v1/assistants)/([^/]+)$', r'\1/{assistant_id}'), + + # Files API + (r'^(/(?:openai/)?v1/files)/([^/]+)(/content)$', r'\1/{file_id}\3'), + (r'^(/(?:openai/)?v1/files)/([^/]+)$', r'\1/{file_id}'), + + # Batches API + (r'^(/(?:openai/)?v1/batches)/([^/]+)(/cancel)$', r'\1/{batch_id}\3'), + (r'^(/(?:openai/)?v1/batches)/([^/]+)$', r'\1/{batch_id}'), + + # Fine-tuning API + (r'^(/(?:openai/)?v1/fine_tuning/jobs)/([^/]+)(/events)$', r'\1/{fine_tuning_job_id}\3'), + (r'^(/(?:openai/)?v1/fine_tuning/jobs)/([^/]+)(/cancel)$', r'\1/{fine_tuning_job_id}\3'), + (r'^(/(?:openai/)?v1/fine_tuning/jobs)/([^/]+)(/checkpoints)$', r'\1/{fine_tuning_job_id}\3'), + (r'^(/(?:openai/)?v1/fine_tuning/jobs)/([^/]+)$', r'\1/{fine_tuning_job_id}'), + + # Models API + (r'^(/(?:openai/)?v1/models)/([^/]+)$', r'\1/{model}'), + ] + + # Apply patterns in order + for pattern, replacement in patterns: + normalized = re.sub(pattern, replacement, route) + if normalized != route: + return normalized + + # Return original route if no pattern matched + return route + + async def check_if_request_size_is_safe(request: Request) -> bool: """ Enterprise Only: diff --git a/litellm/proxy/auth/user_api_key_auth.py b/litellm/proxy/auth/user_api_key_auth.py index bc0c164a0ad..7e7c7c8c90c 100644 --- a/litellm/proxy/auth/user_api_key_auth.py +++ b/litellm/proxy/auth/user_api_key_auth.py @@ -28,8 +28,8 @@ from litellm.proxy.auth.auth_checks import ( _delete_cache_key_object, _get_user_role, _is_user_proxy_admin, - _virtual_key_max_budget_check, _virtual_key_max_budget_alert_check, + _virtual_key_max_budget_check, _virtual_key_soft_budget_check, can_key_call_model, common_checks, @@ -45,6 +45,7 @@ from litellm.proxy.auth.auth_utils import ( get_end_user_id_from_request_body, get_model_from_request, get_request_route, + normalize_request_route, pre_db_read_auth_checks, route_in_additonal_public_routes, ) @@ -1261,7 +1262,7 @@ async def user_api_key_auth( if end_user_id is not None: user_api_key_auth_obj.end_user_id = end_user_id - user_api_key_auth_obj.request_route = route + user_api_key_auth_obj.request_route = normalize_request_route(route) return user_api_key_auth_obj diff --git a/tests/test_litellm/integrations/test_prometheus_labels.py b/tests/test_litellm/integrations/test_prometheus_labels.py index 8a4295e98fc..c0b863ef6ee 100644 --- a/tests/test_litellm/integrations/test_prometheus_labels.py +++ b/tests/test_litellm/integrations/test_prometheus_labels.py @@ -3,7 +3,7 @@ Unit tests for prometheus metric labels configuration """ from litellm.types.integrations.prometheus import ( PrometheusMetricLabels, - UserAPIKeyLabelNames + UserAPIKeyLabelNames, ) @@ -42,9 +42,10 @@ def test_user_email_label_exists(): def test_prometheus_metric_labels_structure(): """Test that all required prometheus metrics have proper label structure""" - from litellm.types.integrations.prometheus import DEFINED_PROMETHEUS_METRICS from typing import get_args + from litellm.types.integrations.prometheus import DEFINED_PROMETHEUS_METRICS + # Test a few key metrics to ensure they have proper label structure test_metrics = [ "litellm_proxy_total_requests_metric", @@ -69,8 +70,161 @@ def test_prometheus_metric_labels_structure(): print(f"✅ {metric_name} has proper label structure with user_email") +def test_route_normalization_for_responses_api(): + """ + Test that route normalization prevents high cardinality in Prometheus metrics + for the /v1/responses/{response_id} endpoint. + + Issue: https://github.com/BerriAI/litellm/issues/XXXX + Each unique response ID was creating a separate metric line, causing the + /metrics endpoint to grow to ~30MB and take ~40 seconds to respond. + + Fix: Routes are normalized to collapse dynamic IDs into placeholders. + """ + from litellm.proxy.auth.auth_utils import normalize_request_route + + # Test responses API routes + responses_routes = [ + ("/v1/responses/1234567890", "/v1/responses/{response_id}"), + ("/v1/responses/9876543210", "/v1/responses/{response_id}"), + ("/v1/responses/abcdefghij", "/v1/responses/{response_id}"), + ("/v1/responses/resp_abc123", "/v1/responses/{response_id}"), + ("/v1/responses/litellm_poll_xyz", "/v1/responses/{response_id}"), + ] + + for original, expected in responses_routes: + normalized = normalize_request_route(original) + assert normalized == expected, \ + f"Failed: {original} -> {normalized} (expected {expected})" + + # Verify cardinality reduction + unique_normalized = set(normalize_request_route(route) for route, _ in responses_routes) + assert len(unique_normalized) == 1, \ + f"Expected 1 unique normalized route, got {len(unique_normalized)}: {unique_normalized}" + + print(f"✅ Responses API routes: {len(responses_routes)} different IDs normalized to 1 metric label") + + +def test_route_normalization_for_sub_routes(): + """Test that sub-routes like /cancel and /input_items are normalized correctly""" + from litellm.proxy.auth.auth_utils import normalize_request_route + + sub_routes = [ + ("/v1/responses/id1/cancel", "/v1/responses/{response_id}/cancel"), + ("/v1/responses/id2/cancel", "/v1/responses/{response_id}/cancel"), + ("/v1/responses/id3/input_items", "/v1/responses/{response_id}/input_items"), + ("/openai/v1/responses/id4/input_items", "/openai/v1/responses/{response_id}/input_items"), + ] + + for original, expected in sub_routes: + normalized = normalize_request_route(original) + assert normalized == expected, \ + f"Failed: {original} -> {normalized} (expected {expected})" + + print("✅ Sub-routes normalized correctly") + + +def test_route_normalization_preserves_static_routes(): + """Test that static routes are not affected by normalization""" + from litellm.proxy.auth.auth_utils import normalize_request_route + + static_routes = [ + "/chat/completions", + "/v1/chat/completions", + "/v1/embeddings", + "/health", + "/metrics", + "/v1/models", + "/v1/responses", # List endpoint without ID + ] + + for route in static_routes: + normalized = normalize_request_route(route) + assert normalized == route, \ + f"Static route should not be modified: {route} -> {normalized}" + + print(f"✅ {len(static_routes)} static routes preserved") + + +def test_route_normalization_other_dynamic_apis(): + """Test normalization for other OpenAI-compatible APIs with dynamic IDs""" + from litellm.proxy.auth.auth_utils import normalize_request_route + + test_cases = [ + # Threads API + ("/v1/threads/thread_123", "/v1/threads/{thread_id}"), + ("/v1/threads/thread_abc/messages", "/v1/threads/{thread_id}/messages"), + ("/v1/threads/thread_abc/runs/run_123", "/v1/threads/{thread_id}/runs/{run_id}"), + + # Vector Stores API + ("/v1/vector_stores/vs_123", "/v1/vector_stores/{vector_store_id}"), + ("/v1/vector_stores/vs_123/files", "/v1/vector_stores/{vector_store_id}/files"), + + # Assistants API + ("/v1/assistants/asst_123", "/v1/assistants/{assistant_id}"), + + # Files API + ("/v1/files/file_123", "/v1/files/{file_id}"), + ("/v1/files/file_123/content", "/v1/files/{file_id}/content"), + + # Batches API + ("/v1/batches/batch_123", "/v1/batches/{batch_id}"), + ("/v1/batches/batch_123/cancel", "/v1/batches/{batch_id}/cancel"), + ] + + for original, expected in test_cases: + normalized = normalize_request_route(original) + assert normalized == expected, \ + f"Failed: {original} -> {normalized} (expected {expected})" + + print(f"✅ {len(test_cases)} other API routes normalized correctly") + + +def test_prometheus_metrics_use_normalized_routes(): + """ + Test that Prometheus metrics use the normalized route in labels + to prevent high cardinality. + """ + from unittest.mock import MagicMock + + from litellm.integrations.prometheus import ( + PrometheusLogger, + UserAPIKeyLabelValues, + prometheus_label_factory, + ) + + # Create a mock PrometheusLogger + prometheus_logger = MagicMock() + prometheus_logger.get_labels_for_metric = PrometheusLogger.get_labels_for_metric.__get__(prometheus_logger) + + # Test with a normalized route + enum_values = UserAPIKeyLabelValues( + route="/v1/responses/{response_id}", # Normalized route + status_code="200", + requested_model="gpt-4", + ) + + labels = prometheus_label_factory( + supported_enum_labels=prometheus_logger.get_labels_for_metric( + metric_name="litellm_proxy_total_requests_metric" + ), + enum_values=enum_values, + ) + + # Verify the route is normalized in labels + assert labels["route"] == "/v1/responses/{response_id}", \ + f"Expected normalized route in labels, got: {labels.get('route')}" + + print("✅ Prometheus metrics use normalized routes in labels") + + if __name__ == "__main__": test_user_email_in_required_metrics() test_user_email_label_exists() test_prometheus_metric_labels_structure() - print("All prometheus label tests passed!") \ No newline at end of file + test_route_normalization_for_responses_api() + test_route_normalization_for_sub_routes() + test_route_normalization_preserves_static_routes() + test_route_normalization_other_dynamic_apis() + test_prometheus_metrics_use_normalized_routes() + print("\n✅ All prometheus label tests passed!") \ No newline at end of file