mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-17 23:51:30 +00:00
Pre-call limiters reject before a deployment is attached to request_data, so the failure hook could not resolve api_provider for router aliases and emitted api_provider="None" on litellm_proxy_failed_requests_metric_total and litellm_proxy_total_requests_metric_total. Fall back to the provider the limiter already resolved onto RateLimitError.llm_provider, keeping request data as the first source and ignoring the proxy placeholder. Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
803 lines
29 KiB
Python
803 lines
29 KiB
Python
"""
|
|
Unit tests for prometheus metric labels configuration
|
|
"""
|
|
|
|
import pytest
|
|
|
|
from litellm.types.integrations.prometheus import (
|
|
PrometheusMetricLabels,
|
|
UserAPIKeyLabelNames,
|
|
)
|
|
|
|
|
|
def _clear_prometheus_registry() -> None:
|
|
from prometheus_client import REGISTRY
|
|
|
|
for collector in list(REGISTRY._collector_to_names.keys()):
|
|
try:
|
|
REGISTRY.unregister(collector)
|
|
except Exception:
|
|
pass
|
|
|
|
|
|
def _collected_samples(metric_name: str):
|
|
from prometheus_client import REGISTRY
|
|
|
|
return [
|
|
sample
|
|
for metric in REGISTRY.collect()
|
|
for sample in metric.samples
|
|
if sample.name == metric_name
|
|
]
|
|
|
|
|
|
def test_user_email_in_required_metrics():
|
|
"""
|
|
Test that user_email label is present in all the metrics that should have it:
|
|
- litellm_proxy_total_requests_metric (already had it)
|
|
- litellm_proxy_failed_requests_metric (added)
|
|
- litellm_input_tokens_metric (added)
|
|
- litellm_output_tokens_metric (added)
|
|
- litellm_requests_metric (already had it)
|
|
- litellm_spend_metric (added)
|
|
"""
|
|
user_email_label = UserAPIKeyLabelNames.USER_EMAIL.value
|
|
|
|
# Metrics that should have user_email
|
|
metrics_with_user_email = [
|
|
"litellm_proxy_total_requests_metric",
|
|
"litellm_proxy_failed_requests_metric",
|
|
"litellm_input_tokens_metric",
|
|
"litellm_output_tokens_metric",
|
|
"litellm_requests_metric",
|
|
"litellm_spend_metric",
|
|
]
|
|
|
|
for metric_name in metrics_with_user_email:
|
|
labels = PrometheusMetricLabels.get_labels(metric_name)
|
|
assert (
|
|
user_email_label in labels
|
|
), f"Metric {metric_name} should contain user_email label"
|
|
print(f"✅ {metric_name} contains user_email label")
|
|
|
|
|
|
def test_model_id_in_extended_metric_set():
|
|
"""
|
|
Test that model_id label is present in all the metrics that should have it
|
|
"""
|
|
model_id_label = UserAPIKeyLabelNames.MODEL_ID.value
|
|
|
|
# Metrics that should have model_id
|
|
metrics_with_model_id = [
|
|
"litellm_proxy_total_requests_metric",
|
|
"litellm_proxy_failed_requests_metric",
|
|
"litellm_input_tokens_metric",
|
|
"litellm_output_tokens_metric",
|
|
"litellm_requests_metric",
|
|
"litellm_spend_metric",
|
|
"litellm_llm_api_latency_metric",
|
|
"litellm_remaining_requests_metric",
|
|
"litellm_deployment_successful_fallbacks",
|
|
"litellm_cache_hits_metric",
|
|
"litellm_cache_misses_metric",
|
|
"litellm_remaining_api_key_requests_for_model",
|
|
"litellm_remaining_api_key_tokens_for_model",
|
|
"litellm_llm_api_failed_requests_metric",
|
|
]
|
|
|
|
for metric_name in metrics_with_model_id:
|
|
labels = PrometheusMetricLabels.get_labels(metric_name)
|
|
assert (
|
|
model_id_label in labels
|
|
), f"Metric {metric_name} should contain model_id label"
|
|
print(f"✅ {metric_name} contains model_id label")
|
|
|
|
|
|
def test_api_provider_in_spend_and_requests_metrics():
|
|
"""
|
|
Test that api_provider label is present in spend and requests metrics
|
|
so users can build spend-by-provider and request-count-by-provider dashboards.
|
|
"""
|
|
api_provider_label = UserAPIKeyLabelNames.API_PROVIDER.value
|
|
|
|
for metric_name in ["litellm_spend_metric", "litellm_requests_metric"]:
|
|
labels = PrometheusMetricLabels.get_labels(metric_name)
|
|
assert (
|
|
api_provider_label in labels
|
|
), f"Metric {metric_name} should contain api_provider label"
|
|
print(f"✅ {metric_name} contains api_provider label")
|
|
|
|
|
|
def test_api_provider_in_token_latency_and_request_metrics():
|
|
"""
|
|
Regression test for LIT-4178.
|
|
|
|
These metrics are all emitted from the same call site (async_log_success_event
|
|
/ async_post_call_failure_hook) as litellm_spend_metric and
|
|
litellm_requests_metric, which already carry api_provider. They were the odd
|
|
ones out with no way to break down tokens, latency, request counts or cache
|
|
hits by upstream provider, even though the provider is available on the
|
|
payload as custom_llm_provider.
|
|
"""
|
|
api_provider_label = UserAPIKeyLabelNames.API_PROVIDER.value
|
|
|
|
metrics_with_api_provider = [
|
|
"litellm_llm_api_latency_metric",
|
|
"litellm_llm_api_time_to_first_token_metric",
|
|
"litellm_request_total_latency_metric",
|
|
"litellm_request_queue_time_seconds",
|
|
"litellm_proxy_total_requests_metric",
|
|
"litellm_proxy_failed_requests_metric",
|
|
"litellm_input_tokens_metric",
|
|
"litellm_total_tokens_metric",
|
|
"litellm_output_tokens_metric",
|
|
"litellm_cache_hits_metric",
|
|
"litellm_cache_misses_metric",
|
|
# The remaining cache metrics share _cache_metric_labels, so they pick up
|
|
# api_provider from the same change. Assert them explicitly so the shared
|
|
# list can't silently drop the label from them.
|
|
"litellm_cached_tokens_metric",
|
|
"litellm_provider_cache_read_input_tokens_metric",
|
|
"litellm_provider_cache_creation_input_tokens_metric",
|
|
]
|
|
|
|
for metric_name in metrics_with_api_provider:
|
|
labels = PrometheusMetricLabels.get_labels(metric_name)
|
|
assert (
|
|
api_provider_label in labels
|
|
), f"Metric {metric_name} should contain api_provider label"
|
|
|
|
|
|
def test_api_provider_value_flows_through_label_factory():
|
|
"""
|
|
The label being in the allow-list is necessary but not sufficient: the
|
|
factory must also carry the value from the enum through to the emitted
|
|
label. This would fail if the label were dropped from the metric's list or
|
|
if the value plumbing regressed, which the allow-list assertion above cannot
|
|
catch on its own.
|
|
"""
|
|
from unittest.mock import MagicMock
|
|
|
|
from litellm.integrations.prometheus import (
|
|
PrometheusLogger,
|
|
UserAPIKeyLabelValues,
|
|
prometheus_label_factory,
|
|
)
|
|
|
|
prometheus_logger = MagicMock()
|
|
prometheus_logger._cached_metric_labels = {}
|
|
prometheus_logger.label_filters = {}
|
|
prometheus_logger.get_labels_for_metric = (
|
|
PrometheusLogger.get_labels_for_metric.__get__(prometheus_logger)
|
|
)
|
|
|
|
enum_values = UserAPIKeyLabelValues(
|
|
api_provider="anthropic",
|
|
litellm_model_name="claude-sonnet-4",
|
|
requested_model="claude",
|
|
status_code="200",
|
|
)
|
|
|
|
for metric_name in [
|
|
"litellm_input_tokens_metric",
|
|
"litellm_total_tokens_metric",
|
|
"litellm_output_tokens_metric",
|
|
"litellm_llm_api_latency_metric",
|
|
"litellm_request_total_latency_metric",
|
|
"litellm_proxy_total_requests_metric",
|
|
"litellm_cache_hits_metric",
|
|
]:
|
|
labels = prometheus_label_factory(
|
|
supported_enum_labels=prometheus_logger.get_labels_for_metric(
|
|
metric_name=metric_name
|
|
),
|
|
enum_values=enum_values,
|
|
)
|
|
assert (
|
|
labels.get("api_provider") == "anthropic"
|
|
), f"{metric_name} should emit api_provider=anthropic, got {labels.get('api_provider')!r}"
|
|
|
|
|
|
def test_extract_api_provider_from_request_data_failure_path():
|
|
"""
|
|
On the client-side failure path the provider is not always known. Prefer the
|
|
resolved custom_llm_provider on litellm_params, fall back to a partial
|
|
standard_logging_object (e.g. a stream that broke mid-flight), then infer it
|
|
from the requested model name, and return None only when nothing maps so the
|
|
label emits empty rather than a guess.
|
|
"""
|
|
from litellm.integrations.prometheus import PrometheusLogger
|
|
|
|
extract = PrometheusLogger._extract_api_provider_from_request_data
|
|
|
|
assert extract({"litellm_params": {"custom_llm_provider": "bedrock"}}) == "bedrock"
|
|
assert (
|
|
extract({"standard_logging_object": {"custom_llm_provider": "vertex_ai"}})
|
|
== "vertex_ai"
|
|
)
|
|
# litellm_params wins over standard_logging_object when both are present
|
|
assert (
|
|
extract(
|
|
{
|
|
"litellm_params": {"custom_llm_provider": "openai"},
|
|
"standard_logging_object": {"custom_llm_provider": "azure"},
|
|
}
|
|
)
|
|
== "openai"
|
|
)
|
|
# Fallback: infer provider from the requested model name when the proxy's
|
|
# failure request_data carries only the client-supplied model. This is the
|
|
# common client-side failure case (e.g. an invalid param rejected with 400).
|
|
assert extract({"model": "gpt-4o-mini"}) == "openai"
|
|
assert extract({"litellm_params": {"model": "anthropic/claude-haiku-4-5"}}) == "anthropic"
|
|
assert extract({}) is None
|
|
# Unmappable model name -> None, not a guess
|
|
assert extract({"model": "some-unknown-model-xyz"}) is None
|
|
|
|
|
|
def test_extract_api_provider_swallows_unknown_model_but_logs_unexpected_errors():
|
|
"""
|
|
get_llm_provider raises BadRequestError for a model that maps to no
|
|
provider; that is the expected miss and must resolve to None quietly. Any
|
|
other exception is unexpected (e.g. a real bug) and must not vanish
|
|
silently, nor break metric emission: it is logged and still returns None.
|
|
"""
|
|
from unittest.mock import patch
|
|
|
|
import litellm
|
|
from litellm.integrations.prometheus import PrometheusLogger
|
|
|
|
extract = PrometheusLogger._extract_api_provider_from_request_data
|
|
|
|
# Expected miss: BadRequestError -> None, no log noise
|
|
with patch.object(
|
|
litellm,
|
|
"get_llm_provider",
|
|
side_effect=litellm.exceptions.BadRequestError(
|
|
message="no provider", model="x", llm_provider="y"
|
|
),
|
|
):
|
|
with patch("litellm.integrations.prometheus.verbose_logger") as mock_logger:
|
|
assert extract({"model": "x"}) is None
|
|
mock_logger.debug.assert_not_called()
|
|
|
|
# Unexpected error: must be logged and still return None (never raised)
|
|
with patch.object(litellm, "get_llm_provider", side_effect=RuntimeError("boom")):
|
|
with patch("litellm.integrations.prometheus.verbose_logger") as mock_logger:
|
|
assert extract({"model": "x"}) is None
|
|
mock_logger.debug.assert_called_once()
|
|
|
|
|
|
def test_user_email_label_exists():
|
|
"""Test that the USER_EMAIL label is properly defined"""
|
|
assert UserAPIKeyLabelNames.USER_EMAIL.value == "user_email"
|
|
|
|
|
|
def test_prometheus_metric_labels_structure():
|
|
"""Test that all required prometheus metrics have proper label structure"""
|
|
from typing import get_args
|
|
|
|
from litellm.types.integrations.prometheus import DEFINED_PROMETHEUS_METRICS
|
|
|
|
# Test a few key metrics to ensure they have proper label structure
|
|
test_metrics = [
|
|
"litellm_proxy_total_requests_metric",
|
|
"litellm_proxy_failed_requests_metric",
|
|
"litellm_input_tokens_metric",
|
|
"litellm_output_tokens_metric",
|
|
"litellm_spend_metric",
|
|
]
|
|
|
|
for metric_name in test_metrics:
|
|
# Check metric is in DEFINED_PROMETHEUS_METRICS
|
|
assert metric_name in get_args(
|
|
DEFINED_PROMETHEUS_METRICS
|
|
), f"{metric_name} should be in DEFINED_PROMETHEUS_METRICS"
|
|
|
|
# Check labels can be retrieved
|
|
labels = PrometheusMetricLabels.get_labels(metric_name)
|
|
assert isinstance(labels, list), f"Labels for {metric_name} should be a list"
|
|
assert len(labels) > 0, f"Labels for {metric_name} should not be empty"
|
|
|
|
# Check user_email is in the labels
|
|
assert "user_email" in labels, f"{metric_name} should have user_email label"
|
|
|
|
print(f"✅ {metric_name} has proper label structure with user_email")
|
|
|
|
|
|
def test_model_id_in_required_metrics():
|
|
"""
|
|
Test that model_id label is present in all the metrics that should have it:
|
|
- litellm_proxy_total_requests_metric
|
|
- litellm_proxy_failed_requests_metric
|
|
- litellm_request_total_latency_metric
|
|
- litellm_llm_api_time_to_first_token_metric
|
|
"""
|
|
model_id_label = UserAPIKeyLabelNames.MODEL_ID.value
|
|
|
|
# Metrics that should have model_id
|
|
metrics_with_model_id = [
|
|
"litellm_proxy_total_requests_metric",
|
|
"litellm_proxy_failed_requests_metric",
|
|
"litellm_request_total_latency_metric",
|
|
"litellm_llm_api_time_to_first_token_metric",
|
|
]
|
|
|
|
for metric_name in metrics_with_model_id:
|
|
labels = PrometheusMetricLabels.get_labels(metric_name)
|
|
assert (
|
|
model_id_label in labels
|
|
), f"Metric {metric_name} should contain model_id label"
|
|
print(f"✅ {metric_name} contains model_id label")
|
|
|
|
|
|
def test_requested_model_in_spend_and_requests_metrics():
|
|
"""
|
|
Regression test for LIT-3796.
|
|
|
|
litellm_spend_metric and litellm_requests_metric must expose the
|
|
requested_model label so spend and request counts can be grouped by the
|
|
model alias the caller asked for, not just the backend deployment that
|
|
served the request. The sibling token metrics (input/output/total)
|
|
already carry this label; spend and requests were the odd ones out, even
|
|
though both are emitted side-by-side from the same call site.
|
|
"""
|
|
requested_model_label = UserAPIKeyLabelNames.REQUESTED_MODEL.value
|
|
|
|
metrics_with_requested_model = [
|
|
"litellm_spend_metric",
|
|
"litellm_requests_metric",
|
|
"litellm_input_tokens_metric",
|
|
"litellm_output_tokens_metric",
|
|
"litellm_total_tokens_metric",
|
|
]
|
|
|
|
for metric_name in metrics_with_requested_model:
|
|
labels = PrometheusMetricLabels.get_labels(metric_name)
|
|
assert requested_model_label in labels, (
|
|
f"Metric {metric_name} should contain requested_model label"
|
|
)
|
|
|
|
|
|
def test_route_normalization_for_responses_api():
|
|
"""
|
|
Test that route normalization prevents high cardinality in Prometheus metrics
|
|
for the /v1/responses/{response_id} endpoint.
|
|
|
|
Issue: https://github.com/BerriAI/litellm/issues/XXXX
|
|
Each unique response ID was creating a separate metric line, causing the
|
|
/metrics endpoint to grow to ~30MB and take ~40 seconds to respond.
|
|
|
|
Fix: Routes are normalized to collapse dynamic IDs into placeholders.
|
|
"""
|
|
from litellm.proxy.auth.auth_utils import normalize_request_route
|
|
|
|
# Test responses API routes
|
|
responses_routes = [
|
|
("/v1/responses/1234567890", "/v1/responses/{response_id}"),
|
|
("/v1/responses/9876543210", "/v1/responses/{response_id}"),
|
|
("/v1/responses/abcdefghij", "/v1/responses/{response_id}"),
|
|
("/v1/responses/resp_abc123", "/v1/responses/{response_id}"),
|
|
("/v1/responses/litellm_poll_xyz", "/v1/responses/{response_id}"),
|
|
]
|
|
|
|
for original, expected in responses_routes:
|
|
normalized = normalize_request_route(original)
|
|
assert (
|
|
normalized == expected
|
|
), f"Failed: {original} -> {normalized} (expected {expected})"
|
|
|
|
# Verify cardinality reduction
|
|
unique_normalized = set(
|
|
normalize_request_route(route) for route, _ in responses_routes
|
|
)
|
|
assert (
|
|
len(unique_normalized) == 1
|
|
), f"Expected 1 unique normalized route, got {len(unique_normalized)}: {unique_normalized}"
|
|
|
|
print(
|
|
f"✅ Responses API routes: {len(responses_routes)} different IDs normalized to 1 metric label"
|
|
)
|
|
|
|
|
|
def test_route_normalization_for_sub_routes():
|
|
"""Test that sub-routes like /cancel and /input_items are normalized correctly"""
|
|
from litellm.proxy.auth.auth_utils import normalize_request_route
|
|
|
|
sub_routes = [
|
|
("/v1/responses/id1/cancel", "/v1/responses/{response_id}/cancel"),
|
|
("/v1/responses/id2/cancel", "/v1/responses/{response_id}/cancel"),
|
|
("/v1/responses/id3/input_items", "/v1/responses/{response_id}/input_items"),
|
|
(
|
|
"/openai/v1/responses/id4/input_items",
|
|
"/openai/v1/responses/{response_id}/input_items",
|
|
),
|
|
]
|
|
|
|
for original, expected in sub_routes:
|
|
normalized = normalize_request_route(original)
|
|
assert (
|
|
normalized == expected
|
|
), f"Failed: {original} -> {normalized} (expected {expected})"
|
|
|
|
print("✅ Sub-routes normalized correctly")
|
|
|
|
|
|
def test_route_normalization_preserves_static_routes():
|
|
"""Test that static routes are not affected by normalization"""
|
|
from litellm.proxy.auth.auth_utils import normalize_request_route
|
|
|
|
static_routes = [
|
|
"/chat/completions",
|
|
"/v1/chat/completions",
|
|
"/v1/embeddings",
|
|
"/health",
|
|
"/metrics",
|
|
"/v1/models",
|
|
"/v1/responses", # List endpoint without ID
|
|
]
|
|
|
|
for route in static_routes:
|
|
normalized = normalize_request_route(route)
|
|
assert (
|
|
normalized == route
|
|
), f"Static route should not be modified: {route} -> {normalized}"
|
|
|
|
print(f"✅ {len(static_routes)} static routes preserved")
|
|
|
|
|
|
def test_route_normalization_other_dynamic_apis():
|
|
"""Test normalization for other OpenAI-compatible APIs with dynamic IDs"""
|
|
from litellm.proxy.auth.auth_utils import normalize_request_route
|
|
|
|
test_cases = [
|
|
# Threads API
|
|
("/v1/threads/thread_123", "/v1/threads/{thread_id}"),
|
|
("/v1/threads/thread_abc/messages", "/v1/threads/{thread_id}/messages"),
|
|
(
|
|
"/v1/threads/thread_abc/runs/run_123",
|
|
"/v1/threads/{thread_id}/runs/{run_id}",
|
|
),
|
|
# Vector Stores API
|
|
("/v1/vector_stores/vs_123", "/v1/vector_stores/{vector_store_id}"),
|
|
("/v1/vector_stores/vs_123/files", "/v1/vector_stores/{vector_store_id}/files"),
|
|
# Assistants API
|
|
("/v1/assistants/asst_123", "/v1/assistants/{assistant_id}"),
|
|
# Files API
|
|
("/v1/files/file_123", "/v1/files/{file_id}"),
|
|
("/v1/files/file_123/content", "/v1/files/{file_id}/content"),
|
|
# Batches API
|
|
("/v1/batches/batch_123", "/v1/batches/{batch_id}"),
|
|
("/v1/batches/batch_123/cancel", "/v1/batches/{batch_id}/cancel"),
|
|
]
|
|
|
|
for original, expected in test_cases:
|
|
normalized = normalize_request_route(original)
|
|
assert (
|
|
normalized == expected
|
|
), f"Failed: {original} -> {normalized} (expected {expected})"
|
|
|
|
print(f"✅ {len(test_cases)} other API routes normalized correctly")
|
|
|
|
|
|
def test_prometheus_metrics_use_normalized_routes():
|
|
"""
|
|
Test that Prometheus metrics use the normalized route in labels
|
|
to prevent high cardinality.
|
|
"""
|
|
from unittest.mock import MagicMock
|
|
|
|
from litellm.integrations.prometheus import (
|
|
PrometheusLogger,
|
|
UserAPIKeyLabelValues,
|
|
prometheus_label_factory,
|
|
)
|
|
|
|
# Create a mock PrometheusLogger
|
|
prometheus_logger = MagicMock()
|
|
# ``get_labels_for_metric`` reads ``_cached_metric_labels`` and
|
|
# ``label_filters`` off ``self``; default MagicMock attribute access
|
|
# returns Mocks that masquerade as a populated cache, so seed real
|
|
# containers before binding the real method.
|
|
prometheus_logger._cached_metric_labels = {}
|
|
prometheus_logger.label_filters = {}
|
|
prometheus_logger.get_labels_for_metric = (
|
|
PrometheusLogger.get_labels_for_metric.__get__(prometheus_logger)
|
|
)
|
|
|
|
# Test with a normalized route
|
|
enum_values = UserAPIKeyLabelValues(
|
|
route="/v1/responses/{response_id}", # Normalized route
|
|
status_code="200",
|
|
requested_model="gpt-4",
|
|
)
|
|
|
|
labels = prometheus_label_factory(
|
|
supported_enum_labels=prometheus_logger.get_labels_for_metric(
|
|
metric_name="litellm_proxy_total_requests_metric"
|
|
),
|
|
enum_values=enum_values,
|
|
)
|
|
|
|
# Verify the route is normalized in labels
|
|
assert (
|
|
labels["route"] == "/v1/responses/{response_id}"
|
|
), f"Expected normalized route in labels, got: {labels.get('route')}"
|
|
|
|
print("✅ Prometheus metrics use normalized routes in labels")
|
|
|
|
|
|
def test_prometheus_label_value_sanitization():
|
|
"""
|
|
Test that Prometheus label values are sanitized to prevent breaking
|
|
the Prometheus text format.
|
|
|
|
Issue: Unicode Line Separator (U+2028) in label values (e.g. from a
|
|
malformed model name) breaks the Prometheus exposition format, causing
|
|
scrapers like Datadog to fail parsing the entire /metrics endpoint.
|
|
"""
|
|
from litellm.integrations.prometheus import (
|
|
PrometheusLogger,
|
|
UserAPIKeyLabelValues,
|
|
prometheus_label_factory,
|
|
)
|
|
from unittest.mock import MagicMock
|
|
|
|
prometheus_logger = MagicMock()
|
|
prometheus_logger._cached_metric_labels = {}
|
|
prometheus_logger.label_filters = {}
|
|
prometheus_logger.get_labels_for_metric = (
|
|
PrometheusLogger.get_labels_for_metric.__get__(prometheus_logger)
|
|
)
|
|
|
|
# Simulate a model name with U+2028 (Unicode Line Separator) appended
|
|
# and an api_key_alias with newlines and quotes
|
|
enum_values = UserAPIKeyLabelValues(
|
|
requested_model="claude-haiku-4-5-20251001\u2028",
|
|
api_key_alias='My Key "test"\nwith newline',
|
|
route="/v1/chat/completions",
|
|
status_code="400",
|
|
)
|
|
|
|
labels = prometheus_label_factory(
|
|
supported_enum_labels=prometheus_logger.get_labels_for_metric(
|
|
metric_name="litellm_proxy_total_requests_metric"
|
|
),
|
|
enum_values=enum_values,
|
|
)
|
|
|
|
# U+2028 must be stripped
|
|
assert (
|
|
"\u2028" not in labels["requested_model"]
|
|
), f"U+2028 should be removed from label value, got: {repr(labels['requested_model'])}"
|
|
assert labels["requested_model"] == "claude-haiku-4-5-20251001"
|
|
|
|
# Newlines must be replaced with spaces, quotes must be escaped
|
|
assert "\n" not in labels["api_key_alias"]
|
|
assert labels["api_key_alias"] == 'My Key \\"test\\" with newline'
|
|
|
|
print("✅ Prometheus label values are properly sanitized")
|
|
|
|
|
|
def test_prometheus_label_value_sanitization_unicode_paragraph_separator():
|
|
"""Test that U+2029 (Paragraph Separator) is also stripped."""
|
|
from litellm.types.integrations.prometheus import _sanitize_prometheus_label_value
|
|
|
|
result = _sanitize_prometheus_label_value("model\u2029name")
|
|
assert result == "modelname"
|
|
assert "\u2029" not in result
|
|
|
|
print("✅ U+2029 Paragraph Separator is stripped")
|
|
|
|
|
|
def test_prometheus_label_value_sanitization_none():
|
|
"""Test that None values pass through unchanged."""
|
|
from litellm.types.integrations.prometheus import _sanitize_prometheus_label_value
|
|
|
|
assert _sanitize_prometheus_label_value(None) is None
|
|
|
|
print("✅ None values pass through unchanged")
|
|
|
|
|
|
def test_prometheus_label_value_sanitization_non_string_types():
|
|
"""Test that non-string values (int, bool, etc.) are coerced to str."""
|
|
from litellm.types.integrations.prometheus import _sanitize_prometheus_label_value
|
|
|
|
assert _sanitize_prometheus_label_value(200) == "200"
|
|
assert _sanitize_prometheus_label_value(True) == "True"
|
|
assert _sanitize_prometheus_label_value(3.14) == "3.14"
|
|
|
|
print("✅ Non-string values are coerced to str")
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_success_hook_emits_api_provider_value_on_token_metric():
|
|
"""
|
|
End-to-end emit wiring for the success path.
|
|
|
|
The label-list and factory tests prove the label exists and that the factory
|
|
carries a value handed to it, but neither drives the real logger, so deleting
|
|
the production api_provider=standard_logging_payload["custom_llm_provider"]
|
|
assignment in async_log_success_event would still pass them. This drives
|
|
async_log_success_event with a payload whose provider is openai and asserts
|
|
the collected litellm_total_tokens_metric sample actually carries
|
|
api_provider="openai"; it fails if that assignment is removed.
|
|
"""
|
|
import datetime
|
|
|
|
from litellm.integrations.prometheus import PrometheusLogger
|
|
|
|
payload = {
|
|
"id": "t",
|
|
"call_type": "completion",
|
|
"response_cost": 0.001,
|
|
"status": "success",
|
|
"total_tokens": 30,
|
|
"prompt_tokens": 20,
|
|
"completion_tokens": 10,
|
|
"startTime": 1.0,
|
|
"endTime": 2.0,
|
|
"completionStartTime": 1.5,
|
|
"model": "gpt-4o-mini",
|
|
"model_id": "model-123",
|
|
"model_group": "gpt-4o-mini",
|
|
"api_base": "https://api.openai.com",
|
|
"custom_llm_provider": "openai",
|
|
"request_tags": [],
|
|
"end_user": None,
|
|
"cache_hit": False,
|
|
"metadata": {
|
|
"user_api_key_hash": "h",
|
|
"user_api_key_alias": "a",
|
|
"user_api_key_team_id": "t",
|
|
"user_api_key_team_alias": "ta",
|
|
"user_api_key_user_id": "u",
|
|
"user_api_key_user_email": "e@x.com",
|
|
"user_api_key_org_id": None,
|
|
"user_api_key_org_alias": None,
|
|
"requester_metadata": None,
|
|
"user_api_key_end_user_id": None,
|
|
},
|
|
"hidden_params": {"litellm_overhead_time_ms": None, "additional_headers": None},
|
|
}
|
|
|
|
_clear_prometheus_registry()
|
|
try:
|
|
logger = PrometheusLogger()
|
|
now = datetime.datetime.now()
|
|
await logger.async_log_success_event(
|
|
{
|
|
"model": "gpt-4o-mini",
|
|
"litellm_params": {"metadata": {}},
|
|
"standard_logging_object": payload,
|
|
},
|
|
None,
|
|
now,
|
|
now,
|
|
)
|
|
samples = _collected_samples("litellm_total_tokens_metric_total")
|
|
assert samples, "expected litellm_total_tokens_metric to be emitted"
|
|
assert all(s.labels.get("api_provider") == "openai" for s in samples), (
|
|
"collected token metric must carry api_provider=openai, got "
|
|
f"{[s.labels.get('api_provider') for s in samples]}"
|
|
)
|
|
finally:
|
|
_clear_prometheus_registry()
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_failure_hook_emits_api_provider_value_on_failed_requests_metric():
|
|
"""
|
|
End-to-end emit wiring for the failure path.
|
|
|
|
async_post_call_failure_hook on a request whose model resolves to openai must
|
|
emit litellm_proxy_failed_requests_metric with api_provider="openai". This
|
|
fails if the api_provider assignment in the failure hook is removed, which the
|
|
helper-only test cannot catch.
|
|
"""
|
|
from litellm.integrations.prometheus import PrometheusLogger
|
|
from litellm.proxy._types import UserAPIKeyAuth
|
|
|
|
_clear_prometheus_registry()
|
|
try:
|
|
logger = PrometheusLogger()
|
|
await logger.async_post_call_failure_hook(
|
|
request_data={"model": "gpt-4o-mini", "metadata": {}},
|
|
original_exception=Exception("boom"),
|
|
user_api_key_dict=UserAPIKeyAuth(token="tok"),
|
|
)
|
|
samples = _collected_samples("litellm_proxy_failed_requests_metric_total")
|
|
assert samples, "expected litellm_proxy_failed_requests_metric to be emitted"
|
|
assert any(s.labels.get("api_provider") == "openai" for s in samples), (
|
|
"collected failed-requests metric must carry api_provider=openai, got "
|
|
f"{[s.labels.get('api_provider') for s in samples]}"
|
|
)
|
|
finally:
|
|
_clear_prometheus_registry()
|
|
|
|
|
|
async def _failed_requests_api_provider_labels(
|
|
request_data: dict[str, object],
|
|
original_exception: Exception,
|
|
) -> list[str]:
|
|
from litellm.integrations.prometheus import PrometheusLogger
|
|
from litellm.proxy._types import UserAPIKeyAuth
|
|
|
|
_clear_prometheus_registry()
|
|
try:
|
|
await PrometheusLogger().async_post_call_failure_hook(
|
|
request_data=request_data,
|
|
original_exception=original_exception,
|
|
user_api_key_dict=UserAPIKeyAuth(token="tok"),
|
|
)
|
|
return [
|
|
s.labels.get("api_provider")
|
|
for s in _collected_samples("litellm_proxy_failed_requests_metric_total")
|
|
]
|
|
finally:
|
|
_clear_prometheus_registry()
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_failure_hook_emits_api_provider_from_pre_call_rate_limit_error_for_router_alias():
|
|
"""
|
|
Pre-call limiters reject before a deployment lands on request_data and a
|
|
router alias cannot be inferred from its name, so the provider the limiter
|
|
resolved onto the exception is the only source for the label.
|
|
"""
|
|
from litellm.exceptions import RateLimitType
|
|
from litellm.proxy.common_utils.proxy_rate_limit_error import ProxyRateLimitError
|
|
|
|
err = ProxyRateLimitError(
|
|
detail={"error": "rpm exceeded"},
|
|
rate_limit_type=RateLimitType.REQUESTS,
|
|
model="openai/gpt-5.4-mini",
|
|
llm_provider="openai",
|
|
)
|
|
|
|
assert await _failed_requests_api_provider_labels(
|
|
{"model": "team-chat-model", "metadata": {}}, err
|
|
) == ["openai"]
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_failure_hook_leaves_api_provider_unset_when_rate_limiter_could_not_resolve_provider():
|
|
from litellm.proxy.common_utils.proxy_rate_limit_error import ProxyRateLimitError
|
|
|
|
err = ProxyRateLimitError(detail={"error": "rpm exceeded"}, model="unknown-alias")
|
|
|
|
assert await _failed_requests_api_provider_labels(
|
|
{"model": "unknown-alias", "metadata": {}}, err
|
|
) == ["None"]
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_failure_hook_prefers_request_data_provider_over_exception_provider():
|
|
from litellm.exceptions import RateLimitError
|
|
|
|
err = RateLimitError(message="upstream 429", llm_provider="openai", model="gpt-4o")
|
|
|
|
assert await _failed_requests_api_provider_labels(
|
|
{
|
|
"model": "gpt-4o",
|
|
"metadata": {},
|
|
"litellm_params": {"custom_llm_provider": "azure"},
|
|
},
|
|
err,
|
|
) == ["azure"]
|
|
|
|
|
|
if __name__ == "__main__":
|
|
test_user_email_in_required_metrics()
|
|
test_user_email_label_exists()
|
|
test_prometheus_metric_labels_structure()
|
|
test_route_normalization_for_responses_api()
|
|
test_route_normalization_for_sub_routes()
|
|
test_route_normalization_preserves_static_routes()
|
|
test_route_normalization_other_dynamic_apis()
|
|
test_prometheus_metrics_use_normalized_routes()
|
|
test_prometheus_label_value_sanitization()
|
|
test_prometheus_label_value_sanitization_unicode_paragraph_separator()
|
|
test_prometheus_label_value_sanitization_none()
|
|
test_prometheus_label_value_sanitization_non_string_types()
|
|
print("\n✅ All prometheus label tests passed!")
|