mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
Fix for Prometheus Metric Cardinality Issue with /responses Endpoint
This commit is contained in:
parent
ea2e360cb5
commit
5f80e8d5e8
3 changed files with 242 additions and 5 deletions
|
|
@ -311,6 +311,88 @@ def get_request_route(request: Request) -> str:
|
|||
return request.url.path
|
||||
|
||||
|
||||
def normalize_request_route(route: str) -> str:
|
||||
"""
|
||||
Normalize request routes by replacing dynamic path parameters with placeholders.
|
||||
|
||||
This prevents high cardinality in Prometheus metrics by collapsing routes like:
|
||||
- /v1/responses/1234567890 -> /v1/responses/{response_id}
|
||||
- /v1/threads/thread_123 -> /v1/threads/{thread_id}
|
||||
|
||||
Args:
|
||||
route: The request route path
|
||||
|
||||
Returns:
|
||||
Normalized route with dynamic parameters replaced by placeholders
|
||||
|
||||
Examples:
|
||||
>>> normalize_request_route("/v1/responses/abc123")
|
||||
'/v1/responses/{response_id}'
|
||||
>>> normalize_request_route("/v1/responses/abc123/cancel")
|
||||
'/v1/responses/{response_id}/cancel'
|
||||
>>> normalize_request_route("/chat/completions")
|
||||
'/chat/completions'
|
||||
"""
|
||||
# Define patterns for routes with dynamic IDs
|
||||
# Format: (regex_pattern, replacement_template)
|
||||
patterns = [
|
||||
# Responses API - must come before generic patterns
|
||||
(r'^(/(?:openai/)?v1/responses)/([^/]+)(/input_items)$', r'\1/{response_id}\3'),
|
||||
(r'^(/(?:openai/)?v1/responses)/([^/]+)(/cancel)$', r'\1/{response_id}\3'),
|
||||
(r'^(/(?:openai/)?v1/responses)/([^/]+)$', r'\1/{response_id}'),
|
||||
(r'^(/responses)/([^/]+)(/input_items)$', r'\1/{response_id}\3'),
|
||||
(r'^(/responses)/([^/]+)(/cancel)$', r'\1/{response_id}\3'),
|
||||
(r'^(/responses)/([^/]+)$', r'\1/{response_id}'),
|
||||
|
||||
# Threads API
|
||||
(r'^(/(?:openai/)?v1/threads)/([^/]+)(/runs)/([^/]+)(/steps)/([^/]+)$', r'\1/{thread_id}\3/{run_id}\5/{step_id}'),
|
||||
(r'^(/(?:openai/)?v1/threads)/([^/]+)(/runs)/([^/]+)(/steps)$', r'\1/{thread_id}\3/{run_id}\5'),
|
||||
(r'^(/(?:openai/)?v1/threads)/([^/]+)(/runs)/([^/]+)(/cancel)$', r'\1/{thread_id}\3/{run_id}\5'),
|
||||
(r'^(/(?:openai/)?v1/threads)/([^/]+)(/runs)/([^/]+)(/submit_tool_outputs)$', r'\1/{thread_id}\3/{run_id}\5'),
|
||||
(r'^(/(?:openai/)?v1/threads)/([^/]+)(/runs)/([^/]+)$', r'\1/{thread_id}\3/{run_id}'),
|
||||
(r'^(/(?:openai/)?v1/threads)/([^/]+)(/runs)$', r'\1/{thread_id}\3'),
|
||||
(r'^(/(?:openai/)?v1/threads)/([^/]+)(/messages)/([^/]+)$', r'\1/{thread_id}\3/{message_id}'),
|
||||
(r'^(/(?:openai/)?v1/threads)/([^/]+)(/messages)$', r'\1/{thread_id}\3'),
|
||||
(r'^(/(?:openai/)?v1/threads)/([^/]+)$', r'\1/{thread_id}'),
|
||||
|
||||
# Vector Stores API
|
||||
(r'^(/(?:openai/)?v1/vector_stores)/([^/]+)(/files)/([^/]+)$', r'\1/{vector_store_id}\3/{file_id}'),
|
||||
(r'^(/(?:openai/)?v1/vector_stores)/([^/]+)(/files)$', r'\1/{vector_store_id}\3'),
|
||||
(r'^(/(?:openai/)?v1/vector_stores)/([^/]+)(/file_batches)/([^/]+)$', r'\1/{vector_store_id}\3/{batch_id}'),
|
||||
(r'^(/(?:openai/)?v1/vector_stores)/([^/]+)(/file_batches)$', r'\1/{vector_store_id}\3'),
|
||||
(r'^(/(?:openai/)?v1/vector_stores)/([^/]+)$', r'\1/{vector_store_id}'),
|
||||
|
||||
# Assistants API
|
||||
(r'^(/(?:openai/)?v1/assistants)/([^/]+)$', r'\1/{assistant_id}'),
|
||||
|
||||
# Files API
|
||||
(r'^(/(?:openai/)?v1/files)/([^/]+)(/content)$', r'\1/{file_id}\3'),
|
||||
(r'^(/(?:openai/)?v1/files)/([^/]+)$', r'\1/{file_id}'),
|
||||
|
||||
# Batches API
|
||||
(r'^(/(?:openai/)?v1/batches)/([^/]+)(/cancel)$', r'\1/{batch_id}\3'),
|
||||
(r'^(/(?:openai/)?v1/batches)/([^/]+)$', r'\1/{batch_id}'),
|
||||
|
||||
# Fine-tuning API
|
||||
(r'^(/(?:openai/)?v1/fine_tuning/jobs)/([^/]+)(/events)$', r'\1/{fine_tuning_job_id}\3'),
|
||||
(r'^(/(?:openai/)?v1/fine_tuning/jobs)/([^/]+)(/cancel)$', r'\1/{fine_tuning_job_id}\3'),
|
||||
(r'^(/(?:openai/)?v1/fine_tuning/jobs)/([^/]+)(/checkpoints)$', r'\1/{fine_tuning_job_id}\3'),
|
||||
(r'^(/(?:openai/)?v1/fine_tuning/jobs)/([^/]+)$', r'\1/{fine_tuning_job_id}'),
|
||||
|
||||
# Models API
|
||||
(r'^(/(?:openai/)?v1/models)/([^/]+)$', r'\1/{model}'),
|
||||
]
|
||||
|
||||
# Apply patterns in order
|
||||
for pattern, replacement in patterns:
|
||||
normalized = re.sub(pattern, replacement, route)
|
||||
if normalized != route:
|
||||
return normalized
|
||||
|
||||
# Return original route if no pattern matched
|
||||
return route
|
||||
|
||||
|
||||
async def check_if_request_size_is_safe(request: Request) -> bool:
|
||||
"""
|
||||
Enterprise Only:
|
||||
|
|
|
|||
|
|
@ -28,8 +28,8 @@ from litellm.proxy.auth.auth_checks import (
|
|||
_delete_cache_key_object,
|
||||
_get_user_role,
|
||||
_is_user_proxy_admin,
|
||||
_virtual_key_max_budget_check,
|
||||
_virtual_key_max_budget_alert_check,
|
||||
_virtual_key_max_budget_check,
|
||||
_virtual_key_soft_budget_check,
|
||||
can_key_call_model,
|
||||
common_checks,
|
||||
|
|
@ -45,6 +45,7 @@ from litellm.proxy.auth.auth_utils import (
|
|||
get_end_user_id_from_request_body,
|
||||
get_model_from_request,
|
||||
get_request_route,
|
||||
normalize_request_route,
|
||||
pre_db_read_auth_checks,
|
||||
route_in_additonal_public_routes,
|
||||
)
|
||||
|
|
@ -1261,7 +1262,7 @@ async def user_api_key_auth(
|
|||
if end_user_id is not None:
|
||||
user_api_key_auth_obj.end_user_id = end_user_id
|
||||
|
||||
user_api_key_auth_obj.request_route = route
|
||||
user_api_key_auth_obj.request_route = normalize_request_route(route)
|
||||
|
||||
return user_api_key_auth_obj
|
||||
|
||||
|
|
|
|||
|
|
@ -3,7 +3,7 @@ Unit tests for prometheus metric labels configuration
|
|||
"""
|
||||
from litellm.types.integrations.prometheus import (
|
||||
PrometheusMetricLabels,
|
||||
UserAPIKeyLabelNames
|
||||
UserAPIKeyLabelNames,
|
||||
)
|
||||
|
||||
|
||||
|
|
@ -42,9 +42,10 @@ def test_user_email_label_exists():
|
|||
|
||||
def test_prometheus_metric_labels_structure():
|
||||
"""Test that all required prometheus metrics have proper label structure"""
|
||||
from litellm.types.integrations.prometheus import DEFINED_PROMETHEUS_METRICS
|
||||
from typing import get_args
|
||||
|
||||
from litellm.types.integrations.prometheus import DEFINED_PROMETHEUS_METRICS
|
||||
|
||||
# Test a few key metrics to ensure they have proper label structure
|
||||
test_metrics = [
|
||||
"litellm_proxy_total_requests_metric",
|
||||
|
|
@ -69,8 +70,161 @@ def test_prometheus_metric_labels_structure():
|
|||
print(f"✅ {metric_name} has proper label structure with user_email")
|
||||
|
||||
|
||||
def test_route_normalization_for_responses_api():
|
||||
"""
|
||||
Test that route normalization prevents high cardinality in Prometheus metrics
|
||||
for the /v1/responses/{response_id} endpoint.
|
||||
|
||||
Issue: https://github.com/BerriAI/litellm/issues/XXXX
|
||||
Each unique response ID was creating a separate metric line, causing the
|
||||
/metrics endpoint to grow to ~30MB and take ~40 seconds to respond.
|
||||
|
||||
Fix: Routes are normalized to collapse dynamic IDs into placeholders.
|
||||
"""
|
||||
from litellm.proxy.auth.auth_utils import normalize_request_route
|
||||
|
||||
# Test responses API routes
|
||||
responses_routes = [
|
||||
("/v1/responses/1234567890", "/v1/responses/{response_id}"),
|
||||
("/v1/responses/9876543210", "/v1/responses/{response_id}"),
|
||||
("/v1/responses/abcdefghij", "/v1/responses/{response_id}"),
|
||||
("/v1/responses/resp_abc123", "/v1/responses/{response_id}"),
|
||||
("/v1/responses/litellm_poll_xyz", "/v1/responses/{response_id}"),
|
||||
]
|
||||
|
||||
for original, expected in responses_routes:
|
||||
normalized = normalize_request_route(original)
|
||||
assert normalized == expected, \
|
||||
f"Failed: {original} -> {normalized} (expected {expected})"
|
||||
|
||||
# Verify cardinality reduction
|
||||
unique_normalized = set(normalize_request_route(route) for route, _ in responses_routes)
|
||||
assert len(unique_normalized) == 1, \
|
||||
f"Expected 1 unique normalized route, got {len(unique_normalized)}: {unique_normalized}"
|
||||
|
||||
print(f"✅ Responses API routes: {len(responses_routes)} different IDs normalized to 1 metric label")
|
||||
|
||||
|
||||
def test_route_normalization_for_sub_routes():
|
||||
"""Test that sub-routes like /cancel and /input_items are normalized correctly"""
|
||||
from litellm.proxy.auth.auth_utils import normalize_request_route
|
||||
|
||||
sub_routes = [
|
||||
("/v1/responses/id1/cancel", "/v1/responses/{response_id}/cancel"),
|
||||
("/v1/responses/id2/cancel", "/v1/responses/{response_id}/cancel"),
|
||||
("/v1/responses/id3/input_items", "/v1/responses/{response_id}/input_items"),
|
||||
("/openai/v1/responses/id4/input_items", "/openai/v1/responses/{response_id}/input_items"),
|
||||
]
|
||||
|
||||
for original, expected in sub_routes:
|
||||
normalized = normalize_request_route(original)
|
||||
assert normalized == expected, \
|
||||
f"Failed: {original} -> {normalized} (expected {expected})"
|
||||
|
||||
print("✅ Sub-routes normalized correctly")
|
||||
|
||||
|
||||
def test_route_normalization_preserves_static_routes():
|
||||
"""Test that static routes are not affected by normalization"""
|
||||
from litellm.proxy.auth.auth_utils import normalize_request_route
|
||||
|
||||
static_routes = [
|
||||
"/chat/completions",
|
||||
"/v1/chat/completions",
|
||||
"/v1/embeddings",
|
||||
"/health",
|
||||
"/metrics",
|
||||
"/v1/models",
|
||||
"/v1/responses", # List endpoint without ID
|
||||
]
|
||||
|
||||
for route in static_routes:
|
||||
normalized = normalize_request_route(route)
|
||||
assert normalized == route, \
|
||||
f"Static route should not be modified: {route} -> {normalized}"
|
||||
|
||||
print(f"✅ {len(static_routes)} static routes preserved")
|
||||
|
||||
|
||||
def test_route_normalization_other_dynamic_apis():
|
||||
"""Test normalization for other OpenAI-compatible APIs with dynamic IDs"""
|
||||
from litellm.proxy.auth.auth_utils import normalize_request_route
|
||||
|
||||
test_cases = [
|
||||
# Threads API
|
||||
("/v1/threads/thread_123", "/v1/threads/{thread_id}"),
|
||||
("/v1/threads/thread_abc/messages", "/v1/threads/{thread_id}/messages"),
|
||||
("/v1/threads/thread_abc/runs/run_123", "/v1/threads/{thread_id}/runs/{run_id}"),
|
||||
|
||||
# Vector Stores API
|
||||
("/v1/vector_stores/vs_123", "/v1/vector_stores/{vector_store_id}"),
|
||||
("/v1/vector_stores/vs_123/files", "/v1/vector_stores/{vector_store_id}/files"),
|
||||
|
||||
# Assistants API
|
||||
("/v1/assistants/asst_123", "/v1/assistants/{assistant_id}"),
|
||||
|
||||
# Files API
|
||||
("/v1/files/file_123", "/v1/files/{file_id}"),
|
||||
("/v1/files/file_123/content", "/v1/files/{file_id}/content"),
|
||||
|
||||
# Batches API
|
||||
("/v1/batches/batch_123", "/v1/batches/{batch_id}"),
|
||||
("/v1/batches/batch_123/cancel", "/v1/batches/{batch_id}/cancel"),
|
||||
]
|
||||
|
||||
for original, expected in test_cases:
|
||||
normalized = normalize_request_route(original)
|
||||
assert normalized == expected, \
|
||||
f"Failed: {original} -> {normalized} (expected {expected})"
|
||||
|
||||
print(f"✅ {len(test_cases)} other API routes normalized correctly")
|
||||
|
||||
|
||||
def test_prometheus_metrics_use_normalized_routes():
|
||||
"""
|
||||
Test that Prometheus metrics use the normalized route in labels
|
||||
to prevent high cardinality.
|
||||
"""
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
from litellm.integrations.prometheus import (
|
||||
PrometheusLogger,
|
||||
UserAPIKeyLabelValues,
|
||||
prometheus_label_factory,
|
||||
)
|
||||
|
||||
# Create a mock PrometheusLogger
|
||||
prometheus_logger = MagicMock()
|
||||
prometheus_logger.get_labels_for_metric = PrometheusLogger.get_labels_for_metric.__get__(prometheus_logger)
|
||||
|
||||
# Test with a normalized route
|
||||
enum_values = UserAPIKeyLabelValues(
|
||||
route="/v1/responses/{response_id}", # Normalized route
|
||||
status_code="200",
|
||||
requested_model="gpt-4",
|
||||
)
|
||||
|
||||
labels = prometheus_label_factory(
|
||||
supported_enum_labels=prometheus_logger.get_labels_for_metric(
|
||||
metric_name="litellm_proxy_total_requests_metric"
|
||||
),
|
||||
enum_values=enum_values,
|
||||
)
|
||||
|
||||
# Verify the route is normalized in labels
|
||||
assert labels["route"] == "/v1/responses/{response_id}", \
|
||||
f"Expected normalized route in labels, got: {labels.get('route')}"
|
||||
|
||||
print("✅ Prometheus metrics use normalized routes in labels")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
test_user_email_in_required_metrics()
|
||||
test_user_email_label_exists()
|
||||
test_prometheus_metric_labels_structure()
|
||||
print("All prometheus label tests passed!")
|
||||
test_route_normalization_for_responses_api()
|
||||
test_route_normalization_for_sub_routes()
|
||||
test_route_normalization_preserves_static_routes()
|
||||
test_route_normalization_other_dynamic_apis()
|
||||
test_prometheus_metrics_use_normalized_routes()
|
||||
print("\n✅ All prometheus label tests passed!")
|
||||
Loading…
Add table
Reference in a new issue