diff --git a/cookbook/litellm_proxy_server/grafana_dashboard/dashboard_all_metrics/grafana_dashboard.json b/cookbook/litellm_proxy_server/grafana_dashboard/dashboard_all_metrics/grafana_dashboard.json index 5bd7ed97a55..9dd8daa5d14 100644 --- a/cookbook/litellm_proxy_server/grafana_dashboard/dashboard_all_metrics/grafana_dashboard.json +++ b/cookbook/litellm_proxy_server/grafana_dashboard/dashboard_all_metrics/grafana_dashboard.json @@ -6381,6 +6381,133 @@ ], "title": "litellm_zero_cost_requests rate", "type": "timeseries" + }, + { + "collapsed": false, + "gridPos": { + "h": 1, + "w": 24, + "x": 0, + "y": 438 + }, + "id": 112, + "panels": [], + "title": "Project model rate limits", + "type": "row" + }, + { + "datasource": { + "type": "prometheus", + "uid": "${DS_PROMETHEUS}" + }, + "description": "Configured rate limit for the Project on the requested model in the current window, by rate_limit_type", + "fieldConfig": { + "defaults": { + "color": { + "mode": "palette-classic" + }, + "custom": { + "drawStyle": "line", + "fillOpacity": 10, + "lineWidth": 1, + "showPoints": "never", + "spanNulls": false + }, + "unit": "short" + }, + "overrides": [] + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 439 + }, + "id": 113, + "options": { + "legend": { + "calcs": [], + "displayMode": "list", + "placement": "bottom", + "showLegend": true + }, + "tooltip": { + "mode": "multi", + "sort": "desc" + } + }, + "targets": [ + { + "datasource": { + "type": "prometheus", + "uid": "${DS_PROMETHEUS}" + }, + "editorMode": "code", + "expr": "max by (project_id, project_alias, requested_model, rate_limit_type) (litellm_project_model_rate_limit_allowed_metric)", + "legendFormat": "{{project_alias}} ({{project_id}}) / {{requested_model}} / {{rate_limit_type}}", + "range": true, + "refId": "A" + } + ], + "title": "litellm_project_model_rate_limit_allowed_metric", + "type": "timeseries" + }, + { + "datasource": { + "type": "prometheus", + "uid": "${DS_PROMETHEUS}" + }, + "description": "Requests or tokens the Project has consumed on the requested model in the current rate limit window, by rate_limit_type", + "fieldConfig": { + "defaults": { + "color": { + "mode": "palette-classic" + }, + "custom": { + "drawStyle": "line", + "fillOpacity": 10, + "lineWidth": 1, + "showPoints": "never", + "spanNulls": false + }, + "unit": "short" + }, + "overrides": [] + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 439 + }, + "id": 114, + "options": { + "legend": { + "calcs": [], + "displayMode": "list", + "placement": "bottom", + "showLegend": true + }, + "tooltip": { + "mode": "multi", + "sort": "desc" + } + }, + "targets": [ + { + "datasource": { + "type": "prometheus", + "uid": "${DS_PROMETHEUS}" + }, + "editorMode": "code", + "expr": "max by (project_id, project_alias, requested_model, rate_limit_type) (litellm_project_model_rate_limit_used_metric)", + "legendFormat": "{{project_alias}} ({{project_id}}) / {{requested_model}} / {{rate_limit_type}}", + "range": true, + "refId": "A" + } + ], + "title": "litellm_project_model_rate_limit_used_metric", + "type": "timeseries" } ], "preload": false, diff --git a/litellm/integrations/prometheus.py b/litellm/integrations/prometheus.py index c7bf291a887..0684a8407d5 100644 --- a/litellm/integrations/prometheus.py +++ b/litellm/integrations/prometheus.py @@ -572,6 +572,24 @@ class PrometheusLogger(CustomLogger): labelnames=self.get_labels_for_metric("litellm_team_rate_limit_used_metric"), ) + self.litellm_project_model_rate_limit_allowed_metric = self._gauge_factory( + "litellm_project_model_rate_limit_allowed_metric", + ( + "Configured rate limit for the Project on the requested model in the current window " + "(model_rpm_limit / model_tpm_limit / model_itpm_limit / model_otpm_limit), by rate_limit_type" + ), + labelnames=self.get_labels_for_metric("litellm_project_model_rate_limit_allowed_metric"), + ) + + self.litellm_project_model_rate_limit_used_metric = self._gauge_factory( + "litellm_project_model_rate_limit_used_metric", + ( + "Requests or tokens the Project has consumed on the requested model in the current rate limit " + "window, by rate_limit_type" + ), + labelnames=self.get_labels_for_metric("litellm_project_model_rate_limit_used_metric"), + ) + ######################################## # LLM API Deployment Metrics / analytics ######################################## @@ -1394,6 +1412,8 @@ class PrometheusLogger(CustomLogger): model_group=standard_logging_payload["model_group"], team=user_api_team, team_alias=user_api_team_alias, + project_id=standard_logging_payload["metadata"].get("user_api_key_project_id"), + project_alias=standard_logging_payload["metadata"].get("user_api_key_project_alias"), org_id=user_api_key_org_id, org_alias=user_api_key_org_alias, user=user_id, @@ -1480,7 +1500,7 @@ class PrometheusLogger(CustomLogger): model_id=enum_values.model_id, ) - self._set_key_and_team_rate_limit_metrics( + self._set_v3_rate_limit_allowed_and_used_metrics( standard_logging_payload=standard_logging_payload, # pyright: ignore[reportArgumentType] # isinstance(dict) above narrows the TypedDict to dict[Unknown, Unknown] enum_values=enum_values, ) @@ -2085,65 +2105,139 @@ class PrometheusLogger(CustomLogger): return None return value - def _set_key_and_team_rate_limit_metrics( + def _set_v3_rate_limit_allowed_and_used_metrics( self, standard_logging_payload: StandardLoggingPayload, enum_values: UserAPIKeyLabelValues, ) -> None: - """ - Export the key-level and team-level RPM / TPM limit and current window - usage from the ``x-ratelimit-{api_key,team}-{limit,remaining}-*`` - headers the v3 rate limiter mirrors into the logging payload. The - limiter already read these counters (from Redis when configured) on - the request path, so no extra store lookup happens here. Descriptors - without a configured limit emit no header, so their series is removed - rather than left at the value from before the limit was dropped. - """ + """Export v3 rate-limit limits and window usage from mirrored logging headers.""" descriptor_gauges: Final[ - tuple[tuple[Literal["api_key", "team"], DEFINED_PROMETHEUS_METRICS, Gauge, Gauge], ...] + tuple[ + tuple[ + Literal[ + "api_key", + "team", + "model_per_project", + "model_per_project_itpm", + "model_per_project_otpm", + ], + Literal["requests", "tokens"], + Literal["requests", "tokens", "input_tokens", "output_tokens"], + DEFINED_PROMETHEUS_METRICS, + Gauge, + Gauge, + ], + ..., + ] ] = ( ( "api_key", + "requests", + "requests", + "litellm_api_key_rate_limit_allowed_metric", + self.litellm_api_key_rate_limit_allowed_metric, + self.litellm_api_key_rate_limit_used_metric, + ), + ( + "api_key", + "tokens", + "tokens", "litellm_api_key_rate_limit_allowed_metric", self.litellm_api_key_rate_limit_allowed_metric, self.litellm_api_key_rate_limit_used_metric, ), ( "team", + "requests", + "requests", "litellm_team_rate_limit_allowed_metric", self.litellm_team_rate_limit_allowed_metric, self.litellm_team_rate_limit_used_metric, ), + ( + "team", + "tokens", + "tokens", + "litellm_team_rate_limit_allowed_metric", + self.litellm_team_rate_limit_allowed_metric, + self.litellm_team_rate_limit_used_metric, + ), + ( + "model_per_project", + "requests", + "requests", + "litellm_project_model_rate_limit_allowed_metric", + self.litellm_project_model_rate_limit_allowed_metric, + self.litellm_project_model_rate_limit_used_metric, + ), + ( + "model_per_project", + "tokens", + "tokens", + "litellm_project_model_rate_limit_allowed_metric", + self.litellm_project_model_rate_limit_allowed_metric, + self.litellm_project_model_rate_limit_used_metric, + ), + ( + "model_per_project_itpm", + "tokens", + "input_tokens", + "litellm_project_model_rate_limit_allowed_metric", + self.litellm_project_model_rate_limit_allowed_metric, + self.litellm_project_model_rate_limit_used_metric, + ), + ( + "model_per_project_otpm", + "tokens", + "output_tokens", + "litellm_project_model_rate_limit_allowed_metric", + self.litellm_project_model_rate_limit_allowed_metric, + self.litellm_project_model_rate_limit_used_metric, + ), ) - for descriptor_key, metric_name, allowed_gauge, used_gauge in descriptor_gauges: - for rate_limit_type in ("requests", "tokens"): - self._set_rate_limit_allowed_and_used_gauges( - standard_logging_payload=standard_logging_payload, - enum_values=enum_values, - descriptor_key=descriptor_key, - metric_name=metric_name, - allowed_gauge=allowed_gauge, - used_gauge=used_gauge, - rate_limit_type=rate_limit_type, - ) + for ( + descriptor_key, + header_rate_limit_type, + rate_limit_type, + metric_name, + allowed_gauge, + used_gauge, + ) in descriptor_gauges: + self._set_rate_limit_allowed_and_used_gauges( + standard_logging_payload=standard_logging_payload, + enum_values=enum_values, + descriptor_key=descriptor_key, + header_rate_limit_type=header_rate_limit_type, + metric_name=metric_name, + allowed_gauge=allowed_gauge, + used_gauge=used_gauge, + rate_limit_type=rate_limit_type, + ) def _set_rate_limit_allowed_and_used_gauges( self, standard_logging_payload: StandardLoggingPayload, enum_values: UserAPIKeyLabelValues, - descriptor_key: Literal["api_key", "team"], + descriptor_key: Literal[ + "api_key", + "team", + "model_per_project", + "model_per_project_itpm", + "model_per_project_otpm", + ], + header_rate_limit_type: Literal["requests", "tokens"], metric_name: DEFINED_PROMETHEUS_METRICS, allowed_gauge: Gauge, used_gauge: Gauge, - rate_limit_type: Literal["requests", "tokens"], + rate_limit_type: Literal["requests", "tokens", "input_tokens", "output_tokens"], ) -> None: limit: Final = self._get_int_from_v3_rate_limit_headers( standard_logging_payload=standard_logging_payload, - header_name=f"x-ratelimit-{descriptor_key}-limit-{rate_limit_type}", + header_name=f"x-ratelimit-{descriptor_key}-limit-{header_rate_limit_type}", ) remaining: Final = self._get_int_from_v3_rate_limit_headers( standard_logging_payload=standard_logging_payload, - header_name=f"x-ratelimit-{descriptor_key}-remaining-{rate_limit_type}", + header_name=f"x-ratelimit-{descriptor_key}-remaining-{header_rate_limit_type}", ) labelled_values: Final = replace(enum_values, rate_limit_type=rate_limit_type) labelnames: Final = self.get_labels_for_metric(metric_name) diff --git a/litellm/types/integrations/prometheus.py b/litellm/types/integrations/prometheus.py index 8f4ad26a4fa..cfa40c4f367 100644 --- a/litellm/types/integrations/prometheus.py +++ b/litellm/types/integrations/prometheus.py @@ -198,6 +198,8 @@ class UserAPIKeyLabelNames(Enum): API_KEY_ALIAS = "api_key_alias" TEAM = "team" TEAM_ALIAS = "team_alias" + PROJECT_ID = "project_id" + PROJECT_ALIAS = "project_alias" REQUESTED_MODEL = REQUESTED_MODEL v1_LITELLM_MODEL_NAME = "model" v2_LITELLM_MODEL_NAME = "litellm_model_name" @@ -297,6 +299,8 @@ DEFINED_PROMETHEUS_METRICS = Literal[ "litellm_api_key_rate_limit_used_metric", "litellm_team_rate_limit_allowed_metric", "litellm_team_rate_limit_used_metric", + "litellm_project_model_rate_limit_allowed_metric", + "litellm_project_model_rate_limit_used_metric", "litellm_llm_api_failed_requests_metric", "litellm_callback_logging_failures_metric", "litellm_in_flight_requests", @@ -837,6 +841,15 @@ class PrometheusMetricLabels: litellm_team_rate_limit_used_metric = litellm_team_rate_limit_allowed_metric + litellm_project_model_rate_limit_allowed_metric: ClassVar[tuple[str, ...]] = ( + UserAPIKeyLabelNames.PROJECT_ID.value, + UserAPIKeyLabelNames.PROJECT_ALIAS.value, + UserAPIKeyLabelNames.REQUESTED_MODEL.value, + UserAPIKeyLabelNames.RATE_LIMIT_TYPE.value, + ) + + litellm_project_model_rate_limit_used_metric = litellm_project_model_rate_limit_allowed_metric + litellm_llm_api_failed_requests_metric = [ UserAPIKeyLabelNames.END_USER.value, UserAPIKeyLabelNames.API_KEY_HASH.value, @@ -1048,6 +1061,8 @@ class UserAPIKeyLabelValues: api_key_alias: str | None = None team: str | None = None team_alias: str | None = None + project_id: str | None = None + project_alias: str | None = None model_group: str | None = None requested_model: str | None = None model: str | None = None diff --git a/tests/unit/integrations/test_prometheus_client_ip_user_agent.py b/tests/unit/integrations/test_prometheus_client_ip_user_agent.py index 004ac4dbffb..8f7923428e7 100644 --- a/tests/unit/integrations/test_prometheus_client_ip_user_agent.py +++ b/tests/unit/integrations/test_prometheus_client_ip_user_agent.py @@ -94,7 +94,7 @@ async def test_async_post_call_success_hook_includes_client_ip_user_agent(): logger._increment_token_metrics = MagicMock() logger._increment_remaining_budget_metrics = AsyncMock() logger._set_virtual_key_rate_limit_metrics = MagicMock() - logger._set_key_and_team_rate_limit_metrics = MagicMock() + logger._set_v3_rate_limit_allowed_and_used_metrics = MagicMock() logger._set_latency_metrics = MagicMock() logger.set_llm_deployment_success_metrics = MagicMock() logger._increment_cache_metrics = MagicMock() diff --git a/tests/unit/integrations/test_prometheus_rate_limit_labels.py b/tests/unit/integrations/test_prometheus_rate_limit_labels.py index bf1d68c7714..e4c71ff7f77 100644 --- a/tests/unit/integrations/test_prometheus_rate_limit_labels.py +++ b/tests/unit/integrations/test_prometheus_rate_limit_labels.py @@ -14,8 +14,10 @@ Covers two follow-up gaps to the unified rate-limit error work: """ from collections.abc import Mapping +from typing import Final from unittest.mock import MagicMock, patch +import litellm import pytest from litellm.exceptions import ( @@ -474,11 +476,13 @@ def test_should_ignore_non_int_v3_header_values(bad_value): ) -KEY_AND_TEAM_RATE_LIMIT_METRICS = ( +RATE_LIMIT_METRICS = ( "litellm_api_key_rate_limit_allowed_metric", "litellm_api_key_rate_limit_used_metric", "litellm_team_rate_limit_allowed_metric", "litellm_team_rate_limit_used_metric", + "litellm_project_model_rate_limit_allowed_metric", + "litellm_project_model_rate_limit_used_metric", ) @@ -534,6 +538,8 @@ def _success_kwargs_with_rate_limit_headers(additional_headers: Mapping[str, obj "user_api_key_alias": "key-alias", "user_api_key_team_id": "team-id", "user_api_key_team_alias": "team-alias", + "user_api_key_project_id": "project-id", + "user_api_key_project_alias": "project-alias", "user_api_key_user_id": "u", "user_api_key_user_email": "e@x.com", "user_api_key_org_id": None, @@ -627,6 +633,68 @@ async def test_should_emit_key_and_team_rate_limit_allowed_and_used_from_v3_head _clear_prometheus_registry() +@pytest.mark.asyncio +async def test_should_emit_project_model_rate_limit_allowed_and_used_from_v3_headers() -> None: + _clear_prometheus_registry() + try: + await _run_success_event( + { + "x-ratelimit-model_per_project-limit-requests": 100, + "x-ratelimit-model_per_project-remaining-requests": 99, + "x-ratelimit-model_per_project-limit-tokens": 10000, + "x-ratelimit-model_per_project-remaining-tokens": 9950, + "x-ratelimit-model_per_project_itpm-limit-tokens": 2000, + "x-ratelimit-model_per_project_itpm-remaining-tokens": 1900, + "x-ratelimit-model_per_project_otpm-limit-tokens": 3000, + "x-ratelimit-model_per_project_otpm-remaining-tokens": 2750, + } + ) + + project_requests: Final = ( + ("project_alias", "project-alias"), + ("project_id", "project-id"), + ("rate_limit_type", "requests"), + ("requested_model", "anthropic-haiku-4-5"), + ) + project_tokens: Final = ( + ("project_alias", "project-alias"), + ("project_id", "project-id"), + ("rate_limit_type", "tokens"), + ("requested_model", "anthropic-haiku-4-5"), + ) + project_input_tokens: Final = ( + ("project_alias", "project-alias"), + ("project_id", "project-id"), + ("rate_limit_type", "input_tokens"), + ("requested_model", "anthropic-haiku-4-5"), + ) + project_output_tokens: Final = ( + ("project_alias", "project-alias"), + ("project_id", "project-id"), + ("rate_limit_type", "output_tokens"), + ("requested_model", "anthropic-haiku-4-5"), + ) + + assert _collected_samples("litellm_project_model_rate_limit_allowed_metric") == { + project_requests: 100, + project_tokens: 10000, + project_input_tokens: 2000, + project_output_tokens: 3000, + } + assert _collected_samples("litellm_project_model_rate_limit_used_metric") == { + project_requests: 1, + project_tokens: 50, + project_input_tokens: 100, + project_output_tokens: 250, + } + assert _collected_samples("litellm_api_key_rate_limit_allowed_metric") == {} + assert _collected_samples("litellm_api_key_rate_limit_used_metric") == {} + assert _collected_samples("litellm_team_rate_limit_allowed_metric") == {} + assert _collected_samples("litellm_team_rate_limit_used_metric") == {} + finally: + _clear_prometheus_registry() + + @pytest.mark.asyncio async def test_should_emit_only_the_dimensions_the_limiter_enforced(): """ @@ -701,6 +769,79 @@ async def test_should_drop_key_and_team_series_once_the_limiter_stops_reporting_ _clear_prometheus_registry() +@pytest.mark.asyncio +async def test_should_drop_project_model_series_once_the_limiter_stops_reporting_a_limit() -> None: + _clear_prometheus_registry() + try: + logger: Final = PrometheusLogger() + await _run_success_event( + { + "x-ratelimit-model_per_project-limit-requests": 100, + "x-ratelimit-model_per_project-remaining-requests": 99, + "x-ratelimit-model_per_project-limit-tokens": 10000, + "x-ratelimit-model_per_project-remaining-tokens": 9950, + "x-ratelimit-model_per_project_itpm-limit-tokens": 2000, + "x-ratelimit-model_per_project_itpm-remaining-tokens": 1900, + "x-ratelimit-model_per_project_otpm-limit-tokens": 3000, + "x-ratelimit-model_per_project_otpm-remaining-tokens": 2750, + }, + logger=logger, + ) + await _run_success_event( + { + "x-ratelimit-model_per_project-limit-requests": 100, + "x-ratelimit-model_per_project-remaining-requests": 96, + }, + logger=logger, + ) + + project_requests: Final = ( + ("project_alias", "project-alias"), + ("project_id", "project-id"), + ("rate_limit_type", "requests"), + ("requested_model", "anthropic-haiku-4-5"), + ) + assert _collected_samples("litellm_project_model_rate_limit_allowed_metric") == { + project_requests: 100, + } + assert _collected_samples("litellm_project_model_rate_limit_used_metric") == { + project_requests: 4, + } + finally: + _clear_prometheus_registry() + + +@pytest.mark.asyncio +async def test_project_model_rate_limit_allowed_uses_same_custom_project_alias_label_as_requests() -> None: + original_custom_labels: Final = litellm.custom_prometheus_metadata_labels + litellm.custom_prometheus_metadata_labels = ["metadata.user_api_key_project_alias"] + _clear_prometheus_registry() + try: + logger: Final = PrometheusLogger() + await _run_success_event( + { + "x-ratelimit-model_per_project-limit-requests": 100, + "x-ratelimit-model_per_project-remaining-requests": 99, + }, + logger=logger, + ) + + allowed_samples: Final = _collected_samples("litellm_project_model_rate_limit_allowed_metric") + request_samples: Final = _collected_samples("litellm_proxy_total_requests_metric_total") + assert len(allowed_samples) == 1 + assert len(request_samples) == 1 + allowed_labels: Final = dict(next(iter(allowed_samples))) + request_labels: Final = dict(next(iter(request_samples))) + assert allowed_labels["metadata_user_api_key_project_alias"] == "project-alias" + assert ( + allowed_labels["metadata_user_api_key_project_alias"] + == request_labels["metadata_user_api_key_project_alias"] + ) + finally: + litellm.custom_prometheus_metadata_labels = original_custom_labels + _clear_prometheus_registry() + + @pytest.mark.asyncio @pytest.mark.parametrize( "additional_headers", @@ -710,16 +851,25 @@ async def test_should_drop_key_and_team_series_once_the_limiter_stops_reporting_ {"x-ratelimit-api_key-limit-requests": 10}, {"x-ratelimit-api_key-limit-requests": "10", "x-ratelimit-api_key-remaining-requests": "7"}, {"x-ratelimit-team-limit-tokens": True, "x-ratelimit-team-remaining-tokens": 5}, + {"x-ratelimit-model_per_project-limit-requests": 10}, + { + "x-ratelimit-model_per_project-limit-requests": "10", + "x-ratelimit-model_per_project-remaining-requests": "7", + }, + { + "x-ratelimit-model_per_project-limit-tokens": True, + "x-ratelimit-model_per_project-remaining-tokens": 5, + }, ], ) -async def test_should_emit_no_key_or_team_rate_limit_series_without_a_complete_int_pair( - additional_headers, -): +async def test_should_emit_no_rate_limit_series_without_a_complete_int_pair( + additional_headers: Mapping[str, object] | None, +) -> None: _clear_prometheus_registry() try: await _run_success_event(additional_headers) - for metric_name in KEY_AND_TEAM_RATE_LIMIT_METRICS: + for metric_name in RATE_LIMIT_METRICS: assert _collected_samples(metric_name) == {}, metric_name finally: _clear_prometheus_registry()