fix(bugbot): hoist provider resolver + opt-in prom rate-limit labels

- dynamic_rate_limiter.py: hoist resolve_llm_provider_for_rate_limit
  above the TPM/RPM if/elif so the lookup runs once per request, matching
  the pattern in dynamic_rate_limiter_v3.py.
- prometheus.py: gate the new rate_limit_category / rate_limit_type
  labels on litellm_proxy_failed_requests_metric behind
  litellm.prometheus_emit_rate_limit_labels (default False). Mirrors the
  existing prometheus_emit_stream_label opt-in. Preserves the metric's
  pre-unification label set so existing dashboards / recording rules
  keep matching after upgrade; operators can enable the new labels once
  downstream consumers include them.
- Tests updated: default-off back-compat case, opt-in path enables the
  flag before asserting label presence.
This commit is contained in:
mateo-berri 2026-05-14 22:15:11 +00:00
parent f3907ea0ce
commit 57682ee792
5 changed files with 95 additions and 42 deletions

View file

@ -416,6 +416,13 @@ custom_prometheus_metadata_labels: List[str] = []
custom_prometheus_tags: List[str] = []
prometheus_metrics_config: Optional[List] = None
prometheus_emit_stream_label: bool = False
# Opt-in: emit `rate_limit_category` and `rate_limit_type` labels on
# `litellm_proxy_failed_requests_metric`. Off by default to preserve the
# pre-unification label set so existing dashboards / recording rules keyed on
# that metric keep matching after upgrade. Enable when downstream consumers
# are ready to split 429s by source (vendor vs. litellm) and dimension
# (RPM/TPM/concurrent/budget).
prometheus_emit_rate_limit_labels: bool = False
prometheus_end_user_metrics_max_series_per_metric: Optional[int] = 10000
prometheus_end_user_metrics_ttl_seconds: Optional[float] = 3600.0
prometheus_end_user_metrics_cleanup_interval_seconds: Optional[float] = 60.0

View file

@ -218,11 +218,11 @@ class _PROXY_DynamicRateLimitHandler(CustomLogger):
) = await self.check_available_usage(
model=data["model"], priority=key_priority
)
resolved_model, llm_provider = resolve_llm_provider_for_rate_limit(
data.get("model")
)
### CHECK TPM ###
if available_tpm is not None and available_tpm == 0:
resolved_model, llm_provider = resolve_llm_provider_for_rate_limit(
data.get("model")
)
raise ProxyRateLimitError(
detail={
"error": "Key={} over available TPM={}. Model TPM={}, Active keys={}".format(
@ -238,9 +238,6 @@ class _PROXY_DynamicRateLimitHandler(CustomLogger):
)
### CHECK RPM ###
elif available_rpm is not None and available_rpm == 0:
resolved_model, llm_provider = resolve_llm_provider_for_rate_limit(
data.get("model")
)
raise ProxyRateLimitError(
detail={
"error": "Key={} over available RPM={}. Model RPM={}, Active keys={}".format(

View file

@ -338,12 +338,10 @@ class PrometheusMetricLabels:
UserAPIKeyLabelNames.USER_EMAIL.value,
UserAPIKeyLabelNames.EXCEPTION_STATUS.value,
UserAPIKeyLabelNames.EXCEPTION_CLASS.value,
# Surfaced from RateLimitError.category / .rate_limit_type when the
# underlying exception is a rate-limit error; ``None`` otherwise. Lets
# dashboards split 429s into vendor vs. litellm and by exceeded
# dimension (RPM/TPM/concurrent/budget) without parsing error text.
UserAPIKeyLabelNames.RATE_LIMIT_CATEGORY.value,
UserAPIKeyLabelNames.RATE_LIMIT_TYPE.value,
# ``rate_limit_category`` / ``rate_limit_type`` are appended in
# ``get_labels()`` when ``litellm.prometheus_emit_rate_limit_labels``
# is True. Kept opt-in so existing dashboards keyed on this metric's
# historical label set keep matching after upgrade.
UserAPIKeyLabelNames.ROUTE.value,
UserAPIKeyLabelNames.CLIENT_IP.value,
UserAPIKeyLabelNames.USER_AGENT.value,
@ -740,6 +738,25 @@ class PrometheusMetricLabels:
):
custom_labels.append(UserAPIKeyLabelNames.STREAM.value)
# Conditionally add unified rate-limit labels to
# litellm_proxy_failed_requests_metric. Off by default so the metric's
# historical label set is preserved across upgrade; enable via
# ``litellm.prometheus_emit_rate_limit_labels`` once downstream
# dashboards include the new labels in their matchers / aggregations.
if (
label_name == "litellm_proxy_failed_requests_metric"
and litellm.prometheus_emit_rate_limit_labels is True
):
for _rate_limit_label in (
UserAPIKeyLabelNames.RATE_LIMIT_CATEGORY.value,
UserAPIKeyLabelNames.RATE_LIMIT_TYPE.value,
):
if (
_rate_limit_label not in default_labels
and _rate_limit_label not in custom_labels
):
custom_labels.append(_rate_limit_label)
if label_name in PrometheusMetricLabels._org_label_metrics:
for label in [
UserAPIKeyLabelNames.ORG_ID.value,

View file

@ -783,6 +783,11 @@ async def test_async_post_call_failure_hook(prometheus_logger):
it should increment the litellm_proxy_failed_requests_metric and litellm_proxy_total_requests_metric
"""
# Opt into the unified rate-limit labels so this test exercises the
# full label set surfaced when `prometheus_emit_rate_limit_labels` is on.
original_emit = litellm.prometheus_emit_rate_limit_labels
litellm.prometheus_emit_rate_limit_labels = True
# Mock the prometheus metrics
prometheus_logger.litellm_proxy_failed_requests_metric = MagicMock()
prometheus_logger.litellm_proxy_total_requests_metric = MagicMock()
@ -804,34 +809,37 @@ async def test_async_post_call_failure_hook(prometheus_logger):
request_route="/chat/completions",
)
# Call the function
await prometheus_logger.async_post_call_failure_hook(
request_data=request_data,
original_exception=original_exception,
user_api_key_dict=user_api_key_dict,
)
try:
# Call the function
await prometheus_logger.async_post_call_failure_hook(
request_data=request_data,
original_exception=original_exception,
user_api_key_dict=user_api_key_dict,
)
# Assert failed requests metric was incremented with correct labels
prometheus_logger.litellm_proxy_failed_requests_metric.labels.assert_called_once_with(
end_user=None,
user="test_user",
user_email=None,
hashed_api_key="test_key",
api_key_alias="test_alias",
team="test_team",
team_alias="test_team_alias",
org_id=None,
org_alias=None,
requested_model="gpt-3.5-turbo",
exception_status="429",
exception_class="Openai.RateLimitError",
rate_limit_category="vendor_rate_limit",
rate_limit_type=None,
route=user_api_key_dict.request_route,
model_id=None,
client_ip=None,
user_agent=None,
)
# Assert failed requests metric was incremented with correct labels
prometheus_logger.litellm_proxy_failed_requests_metric.labels.assert_called_once_with(
end_user=None,
user="test_user",
user_email=None,
hashed_api_key="test_key",
api_key_alias="test_alias",
team="test_team",
team_alias="test_team_alias",
org_id=None,
org_alias=None,
requested_model="gpt-3.5-turbo",
exception_status="429",
exception_class="Openai.RateLimitError",
rate_limit_category="vendor_rate_limit",
rate_limit_type=None,
route=user_api_key_dict.request_route,
model_id=None,
client_ip=None,
user_agent=None,
)
finally:
litellm.prometheus_emit_rate_limit_labels = original_emit
prometheus_logger.litellm_proxy_failed_requests_metric.labels().inc.assert_called_once()
# Assert total requests metric was incremented with correct labels

View file

@ -43,10 +43,34 @@ def test_should_register_rate_limit_label_names_on_enum():
def test_should_include_rate_limit_labels_on_failed_requests_metric():
import litellm
original = litellm.prometheus_emit_rate_limit_labels
try:
litellm.prometheus_emit_rate_limit_labels = True
labels = PrometheusMetricLabels.get_labels(
"litellm_proxy_failed_requests_metric"
)
assert "rate_limit_category" in labels
assert "rate_limit_type" in labels
# These must coexist with the legacy exception labels (back-compat).
assert "exception_class" in labels
assert "exception_status" in labels
finally:
litellm.prometheus_emit_rate_limit_labels = original
def test_should_omit_rate_limit_labels_by_default_for_back_compat():
"""Default-off preserves the metric's historical label set so existing
dashboards / recording rules keyed on `litellm_proxy_failed_requests_metric`
keep matching after upgrade."""
import litellm
assert litellm.prometheus_emit_rate_limit_labels is False
labels = PrometheusMetricLabels.get_labels("litellm_proxy_failed_requests_metric")
assert "rate_limit_category" in labels
assert "rate_limit_type" in labels
# These must coexist with the legacy exception labels (back-compat).
assert "rate_limit_category" not in labels
assert "rate_limit_type" not in labels
# Pre-PR labels must still be present.
assert "exception_class" in labels
assert "exception_status" in labels