mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
feat(prometheus): add requested_model label to spend and requests metrics (#31410)
litellm_spend_metric_total and litellm_requests_metric_total previously exposed only the resolved backend model_id and friendly model name, so operators could not group spend or request counts by the model alias the caller actually asked for when a router fronts multiple deployments behind one name. This adds the existing UserAPIKeyLabelNames.REQUESTED_MODEL to both labelname lists; the value is already populated upstream from standard_logging_payload["model_group"] and flows through the shared _increment_top_level_request_and_spend_metrics call site. The sibling token metrics (input/output/total) already carry the label, so this also restores cross-metric consistency. Resolves LIT-3796
This commit is contained in:
parent
c14329128b
commit
ec4e0146c7
3 changed files with 32 additions and 0 deletions
|
|
@ -409,6 +409,7 @@ class PrometheusMetricLabels:
|
|||
UserAPIKeyLabelNames.USER_EMAIL.value,
|
||||
UserAPIKeyLabelNames.CLIENT_IP.value,
|
||||
UserAPIKeyLabelNames.USER_AGENT.value,
|
||||
UserAPIKeyLabelNames.REQUESTED_MODEL.value,
|
||||
UserAPIKeyLabelNames.MODEL_ID.value,
|
||||
UserAPIKeyLabelNames.API_PROVIDER.value,
|
||||
]
|
||||
|
|
@ -424,6 +425,7 @@ class PrometheusMetricLabels:
|
|||
UserAPIKeyLabelNames.USER_EMAIL.value,
|
||||
UserAPIKeyLabelNames.CLIENT_IP.value,
|
||||
UserAPIKeyLabelNames.USER_AGENT.value,
|
||||
UserAPIKeyLabelNames.REQUESTED_MODEL.value,
|
||||
UserAPIKeyLabelNames.MODEL_ID.value,
|
||||
UserAPIKeyLabelNames.API_PROVIDER.value,
|
||||
]
|
||||
|
|
|
|||
|
|
@ -607,6 +607,7 @@ def test_increment_top_level_request_and_spend_metrics(prometheus_logger):
|
|||
api_provider="openai",
|
||||
client_ip=None,
|
||||
user_agent=None,
|
||||
requested_model=None,
|
||||
)
|
||||
prometheus_logger.litellm_requests_metric.labels().inc.assert_called_once()
|
||||
|
||||
|
|
@ -626,6 +627,7 @@ def test_increment_top_level_request_and_spend_metrics(prometheus_logger):
|
|||
api_provider="openai",
|
||||
client_ip=None,
|
||||
user_agent=None,
|
||||
requested_model=None,
|
||||
)
|
||||
prometheus_logger.litellm_spend_metric.labels().inc.assert_called_once_with(0.1)
|
||||
|
||||
|
|
|
|||
|
|
@ -148,6 +148,34 @@ def test_model_id_in_required_metrics():
|
|||
print(f"✅ {metric_name} contains model_id label")
|
||||
|
||||
|
||||
def test_requested_model_in_spend_and_requests_metrics():
|
||||
"""
|
||||
Regression test for LIT-3796.
|
||||
|
||||
litellm_spend_metric and litellm_requests_metric must expose the
|
||||
requested_model label so spend and request counts can be grouped by the
|
||||
model alias the caller asked for, not just the backend deployment that
|
||||
served the request. The sibling token metrics (input/output/total)
|
||||
already carry this label; spend and requests were the odd ones out, even
|
||||
though both are emitted side-by-side from the same call site.
|
||||
"""
|
||||
requested_model_label = UserAPIKeyLabelNames.REQUESTED_MODEL.value
|
||||
|
||||
metrics_with_requested_model = [
|
||||
"litellm_spend_metric",
|
||||
"litellm_requests_metric",
|
||||
"litellm_input_tokens_metric",
|
||||
"litellm_output_tokens_metric",
|
||||
"litellm_total_tokens_metric",
|
||||
]
|
||||
|
||||
for metric_name in metrics_with_requested_model:
|
||||
labels = PrometheusMetricLabels.get_labels(metric_name)
|
||||
assert requested_model_label in labels, (
|
||||
f"Metric {metric_name} should contain requested_model label"
|
||||
)
|
||||
|
||||
|
||||
def test_route_normalization_for_responses_api():
|
||||
"""
|
||||
Test that route normalization prevents high cardinality in Prometheus metrics
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue