mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
fix(prometheus): add model_group label to end-to-end latency metrics (#44860)
litellm_llm_api_latency_metric, litellm_llm_api_time_to_first_token_metric, litellm_request_total_latency_metric, and litellm_deployment_latency_per_output_token previously carried requested_model/litellm_model_name/model_id but not model_group, so pooled-deployment latency couldn't be grouped by model pool on dashboards -- only the proxy-overhead-only metrics (litellm_overhead_latency_metric and friends) had model_group. All four metrics read enum_values.model_group through the existing prometheus_label_factory plumbing, so no new parameter threading was needed, just the label-list addition. Co-authored-by: ahamedshaik16 <24526479+ahamedshaik16@users.noreply.github.com>
This commit is contained in:
parent
740d0435a8
commit
2c1847f8a2
3 changed files with 195 additions and 0 deletions
|
|
@ -416,6 +416,7 @@ def _resolve_deployment_and_latency_caller_identity_labels(
|
|||
|
||||
class PrometheusMetricLabels:
|
||||
litellm_llm_api_latency_metric = [
|
||||
UserAPIKeyLabelNames.MODEL_GROUP.value,
|
||||
UserAPIKeyLabelNames.v1_LITELLM_MODEL_NAME.value,
|
||||
UserAPIKeyLabelNames.API_KEY_HASH.value,
|
||||
UserAPIKeyLabelNames.API_KEY_ALIAS.value,
|
||||
|
|
@ -430,6 +431,7 @@ class PrometheusMetricLabels:
|
|||
]
|
||||
|
||||
litellm_llm_api_time_to_first_token_metric = [
|
||||
UserAPIKeyLabelNames.MODEL_GROUP.value,
|
||||
UserAPIKeyLabelNames.v1_LITELLM_MODEL_NAME.value,
|
||||
UserAPIKeyLabelNames.API_KEY_HASH.value,
|
||||
UserAPIKeyLabelNames.API_KEY_ALIAS.value,
|
||||
|
|
@ -444,6 +446,7 @@ class PrometheusMetricLabels:
|
|||
]
|
||||
|
||||
litellm_request_total_latency_metric = [
|
||||
UserAPIKeyLabelNames.MODEL_GROUP.value,
|
||||
UserAPIKeyLabelNames.END_USER.value,
|
||||
UserAPIKeyLabelNames.API_KEY_HASH.value,
|
||||
UserAPIKeyLabelNames.API_KEY_ALIAS.value,
|
||||
|
|
@ -516,6 +519,7 @@ class PrometheusMetricLabels:
|
|||
]
|
||||
|
||||
litellm_deployment_latency_per_output_token = [
|
||||
UserAPIKeyLabelNames.MODEL_GROUP.value,
|
||||
UserAPIKeyLabelNames.v2_LITELLM_MODEL_NAME.value,
|
||||
UserAPIKeyLabelNames.MODEL_ID.value,
|
||||
UserAPIKeyLabelNames.API_BASE.value,
|
||||
|
|
|
|||
|
|
@ -433,6 +433,7 @@ def test_set_latency_metrics(prometheus_logger):
|
|||
team_alias="test_team_alias",
|
||||
org_id=None,
|
||||
org_alias=None,
|
||||
model_group="openai-gpt",
|
||||
requested_model="openai-gpt",
|
||||
model="gpt-5-mini",
|
||||
model_id="model-123",
|
||||
|
|
@ -453,6 +454,7 @@ def test_set_latency_metrics(prometheus_logger):
|
|||
team_alias="test_team_alias",
|
||||
org_id=None,
|
||||
org_alias=None,
|
||||
model_group="openai-gpt",
|
||||
requested_model="openai-gpt",
|
||||
model="gpt-5-mini",
|
||||
model_id="model-123",
|
||||
|
|
@ -473,6 +475,7 @@ def test_set_latency_metrics(prometheus_logger):
|
|||
team_alias="test_team_alias",
|
||||
org_id=None,
|
||||
org_alias=None,
|
||||
model_group="openai-gpt",
|
||||
requested_model="openai-gpt",
|
||||
model="gpt-5-mini",
|
||||
model_id="model-123",
|
||||
|
|
@ -1054,6 +1057,7 @@ def test_set_llm_deployment_success_metrics(prometheus_logger):
|
|||
|
||||
# Verify latency per output token metric
|
||||
prometheus_logger.litellm_deployment_latency_per_output_token.labels.assert_called_once_with(
|
||||
model_group="my_custom_model_group",
|
||||
litellm_model_name="gpt-5-mini",
|
||||
model_id="model-123",
|
||||
api_base="https://api.openai.com",
|
||||
|
|
|
|||
|
|
@ -980,6 +980,193 @@ def test_deployment_tpm_rpm_limit_metrics_emit_model_group_from_enum_values():
|
|||
_clear_prometheus_registry()
|
||||
|
||||
|
||||
def test_model_group_in_latency_metrics():
|
||||
"""
|
||||
Test that model_group label is present on the end-to-end / per-call
|
||||
latency metrics needed to build model-group latency dashboards. These
|
||||
metrics previously only carried requested_model, litellm_model_name and
|
||||
model_id, none of which identify the model_group a pooled deployment
|
||||
belongs to -- only the proxy-overhead-only latency metrics
|
||||
(litellm_overhead_latency_metric and friends) carried model_group.
|
||||
"""
|
||||
model_group_label = UserAPIKeyLabelNames.MODEL_GROUP.value
|
||||
|
||||
metrics_with_model_group = [
|
||||
"litellm_llm_api_latency_metric",
|
||||
"litellm_llm_api_time_to_first_token_metric",
|
||||
"litellm_request_total_latency_metric",
|
||||
"litellm_deployment_latency_per_output_token",
|
||||
]
|
||||
|
||||
for metric_name in metrics_with_model_group:
|
||||
labels = PrometheusMetricLabels.get_labels(metric_name)
|
||||
assert (
|
||||
model_group_label in labels
|
||||
), f"Metric {metric_name} should contain model_group label"
|
||||
print(f"✅ {metric_name} contains model_group label")
|
||||
|
||||
|
||||
def test_model_group_value_flows_through_latency_metrics_label_factory():
|
||||
"""
|
||||
The label being in the allow-list is necessary but not sufficient: the
|
||||
factory must also carry the value from the enum through to the emitted
|
||||
label. This would fail if the label were dropped from a metric's list or
|
||||
if the value plumbing regressed, which the allow-list assertion above
|
||||
cannot catch on its own.
|
||||
"""
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
from litellm.integrations.prometheus import (
|
||||
PrometheusLogger,
|
||||
UserAPIKeyLabelValues,
|
||||
prometheus_label_factory,
|
||||
)
|
||||
|
||||
prometheus_logger = MagicMock()
|
||||
prometheus_logger._cached_metric_labels = {}
|
||||
prometheus_logger.label_filters = {}
|
||||
prometheus_logger.get_labels_for_metric = (
|
||||
PrometheusLogger.get_labels_for_metric.__get__(prometheus_logger)
|
||||
)
|
||||
|
||||
enum_values = UserAPIKeyLabelValues(
|
||||
model_group="example-model-group",
|
||||
litellm_model_name="gpt-4o-mini",
|
||||
requested_model="example-model-group",
|
||||
status_code="200",
|
||||
)
|
||||
|
||||
for metric_name in [
|
||||
"litellm_llm_api_latency_metric",
|
||||
"litellm_llm_api_time_to_first_token_metric",
|
||||
"litellm_request_total_latency_metric",
|
||||
"litellm_deployment_latency_per_output_token",
|
||||
]:
|
||||
labels = prometheus_label_factory(
|
||||
supported_enum_labels=prometheus_logger.get_labels_for_metric(
|
||||
metric_name=metric_name
|
||||
),
|
||||
enum_values=enum_values,
|
||||
)
|
||||
assert (
|
||||
labels.get("model_group") == "example-model-group"
|
||||
), f"{metric_name} should emit model_group=example-model-group, got {labels.get('model_group')!r}"
|
||||
|
||||
|
||||
def test_latency_metrics_emit_model_group_from_set_latency_metrics():
|
||||
"""
|
||||
End-to-end emit wiring for _set_latency_metrics.
|
||||
|
||||
The label-list and factory tests above prove the label exists and that
|
||||
the factory carries a value handed to it, but neither drives the real
|
||||
_set_latency_metrics code path, so deleting the production model_group
|
||||
plumbing there would still pass them. This calls it directly with a
|
||||
streaming request (so the time-to-first-token branch also fires) and
|
||||
asserts the real litellm_llm_api_latency_metric,
|
||||
litellm_llm_api_time_to_first_token_metric and
|
||||
litellm_request_total_latency_metric Histogram series actually carry it.
|
||||
"""
|
||||
import datetime
|
||||
|
||||
from litellm.integrations.prometheus import PrometheusLogger, UserAPIKeyLabelValues
|
||||
|
||||
_clear_prometheus_registry()
|
||||
try:
|
||||
logger = PrometheusLogger()
|
||||
start_time = datetime.datetime(2024, 1, 1, 0, 0, 0)
|
||||
api_call_start_time = datetime.datetime(2024, 1, 1, 0, 0, 1)
|
||||
completion_start_time = datetime.datetime(2024, 1, 1, 0, 0, 2)
|
||||
end_time = datetime.datetime(2024, 1, 1, 0, 0, 3)
|
||||
|
||||
enum_values = UserAPIKeyLabelValues(
|
||||
model_group="example-model-group",
|
||||
litellm_model_name="gpt-4o-mini",
|
||||
requested_model="example-model-group",
|
||||
status_code="200",
|
||||
)
|
||||
|
||||
logger._set_latency_metrics(
|
||||
kwargs={
|
||||
"start_time": start_time,
|
||||
"end_time": end_time,
|
||||
"api_call_start_time": api_call_start_time,
|
||||
"completion_start_time": completion_start_time,
|
||||
"stream": True,
|
||||
"litellm_params": {"metadata": {}},
|
||||
},
|
||||
model="gpt-4o-mini",
|
||||
user_api_key=None,
|
||||
user_api_key_alias=None,
|
||||
user_api_team=None,
|
||||
user_api_team_alias=None,
|
||||
enum_values=enum_values,
|
||||
)
|
||||
|
||||
for metric in (
|
||||
logger.litellm_llm_api_latency_metric,
|
||||
logger.litellm_llm_api_time_to_first_token_metric,
|
||||
logger.litellm_request_total_latency_metric,
|
||||
):
|
||||
index = metric._labelnames.index("model_group")
|
||||
values = {sample_key[index] for sample_key in metric._metrics}
|
||||
assert values == {"example-model-group"}, (
|
||||
f"expected model_group=example-model-group on {metric._name}, got {values}"
|
||||
)
|
||||
finally:
|
||||
_clear_prometheus_registry()
|
||||
|
||||
|
||||
def test_deployment_latency_per_output_token_emits_model_group_from_enum_values():
|
||||
"""
|
||||
End-to-end emit wiring for litellm_deployment_latency_per_output_token.
|
||||
|
||||
Drives set_llm_deployment_success_metrics directly (its only caller) with
|
||||
output_tokens > 0 so the latency-per-token branch fires, and asserts the
|
||||
real Histogram series carries model_group; fails if that label-list
|
||||
addition or the enum_values plumbing is removed.
|
||||
"""
|
||||
import datetime
|
||||
|
||||
from litellm.integrations.prometheus import PrometheusLogger, UserAPIKeyLabelValues
|
||||
|
||||
_clear_prometheus_registry()
|
||||
try:
|
||||
logger = PrometheusLogger()
|
||||
start_time = datetime.datetime(2024, 1, 1, 0, 0, 0)
|
||||
end_time = datetime.datetime(2024, 1, 1, 0, 0, 2)
|
||||
enum_values = UserAPIKeyLabelValues(
|
||||
model_group="example-model-group",
|
||||
litellm_model_name="gpt-4o-mini",
|
||||
requested_model="example-model-group",
|
||||
status_code="200",
|
||||
)
|
||||
logger.set_llm_deployment_success_metrics(
|
||||
request_kwargs={
|
||||
"model": "gpt-4o-mini",
|
||||
"litellm_params": {"metadata": {"model_info": {"id": "model-123"}}},
|
||||
"standard_logging_object": {
|
||||
"model_group": "example-model-group",
|
||||
"model_id": "model-123",
|
||||
"api_base": "https://api.openai.com",
|
||||
"hidden_params": {"additional_headers": None, "litellm_overhead_time_ms": None},
|
||||
},
|
||||
},
|
||||
start_time=start_time,
|
||||
end_time=end_time,
|
||||
enum_values=enum_values,
|
||||
output_tokens=10.0,
|
||||
)
|
||||
|
||||
metric = logger.litellm_deployment_latency_per_output_token
|
||||
index = metric._labelnames.index("model_group")
|
||||
values = {sample_key[index] for sample_key in metric._metrics}
|
||||
assert values == {"example-model-group"}, (
|
||||
f"expected model_group=example-model-group on {metric._name}, got {values}"
|
||||
)
|
||||
finally:
|
||||
_clear_prometheus_registry()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
test_user_email_in_required_metrics()
|
||||
test_user_email_label_exists()
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue