mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
(Feat - prometheus) - emit litellm_overhead_latency_metric (#7913)
* add track_llm_api_timing * add track_llm_api_timing * test_litellm_overhead * use ResponseMetadata class for setting hidden params and response overhead * instrument http handler * fix track_llm_api_timing * track_llm_api_timing * emit response overhead on hidden params * fix resp metadata * fix make_sync_openai_embedding_request * test_aaaaatext_completion_endpoint fixes * _get_value_from_hidden_params * set_hidden_params * test_litellm_overhead * test_litellm_overhead * test_litellm_overhead * fix import * test_litellm_overhead_stream * add LiteLLMLoggingObject * use diff folder for testing * use diff folder for overhead testing * test litellm overhead * use typing * clear typing * test_litellm_overhead * fix async_streaming * update_response_metadata * move test file * emit litellm_overhead_latency_metric on prometheus * add prometheus callback * litellm_overhead_latency_metric_bucket * fix apply hidden params * fix StandardLoggingHiddenParams
This commit is contained in:
parent
866fffb50d
commit
4caf4c0277
5 changed files with 37 additions and 8 deletions
|
|
@ -151,7 +151,7 @@ class PrometheusLogger(CustomLogger):
|
|||
|
||||
# Max Budget for Team
|
||||
self.litellm_team_max_budget_metric = Gauge(
|
||||
"litellm_team_max_budget_metric",
|
||||
"litellm_team_max_budget_metric",
|
||||
"Maximum budget set for team",
|
||||
labelnames=["team_id", "team_alias"],
|
||||
)
|
||||
|
|
@ -210,6 +210,20 @@ class PrometheusLogger(CustomLogger):
|
|||
"api_key_alias",
|
||||
],
|
||||
)
|
||||
|
||||
self.litellm_overhead_latency_metric = Histogram(
|
||||
"litellm_overhead_latency_metric",
|
||||
"Latency overhead (milliseconds) added by LiteLLM processing",
|
||||
labelnames=[
|
||||
"model_group",
|
||||
"api_provider",
|
||||
"api_base",
|
||||
"litellm_model_name",
|
||||
"hashed_api_key",
|
||||
"api_key_alias",
|
||||
],
|
||||
buckets=LATENCY_BUCKETS,
|
||||
)
|
||||
# llm api provider budget metrics
|
||||
self.litellm_provider_remaining_budget_metric = Gauge(
|
||||
"litellm_provider_remaining_budget_metric",
|
||||
|
|
@ -325,7 +339,6 @@ class PrometheusLogger(CustomLogger):
|
|||
label_name="litellm_requests_metric"
|
||||
),
|
||||
)
|
||||
|
||||
self._initialize_prometheus_startup_metrics()
|
||||
|
||||
except Exception as e:
|
||||
|
|
@ -988,6 +1001,18 @@ class PrometheusLogger(CustomLogger):
|
|||
"x_ratelimit_remaining_tokens", None
|
||||
)
|
||||
|
||||
if litellm_overhead_time_ms := standard_logging_payload["hidden_params"][
|
||||
"litellm_overhead_time_ms"
|
||||
]:
|
||||
self.litellm_overhead_latency_metric.labels(
|
||||
model_group,
|
||||
llm_provider,
|
||||
api_base,
|
||||
litellm_model_name,
|
||||
standard_logging_payload["metadata"]["user_api_key_hash"],
|
||||
standard_logging_payload["metadata"]["user_api_key_alias"],
|
||||
).observe(litellm_overhead_time_ms)
|
||||
|
||||
if remaining_requests:
|
||||
"""
|
||||
"model_group",
|
||||
|
|
|
|||
|
|
@ -2998,6 +2998,7 @@ class StandardLoggingPayloadSetup:
|
|||
api_base=None,
|
||||
response_cost=None,
|
||||
additional_headers=None,
|
||||
litellm_overhead_time_ms=None,
|
||||
)
|
||||
if hidden_params is not None:
|
||||
for key in StandardLoggingHiddenParams.__annotations__.keys():
|
||||
|
|
@ -3094,6 +3095,7 @@ def get_standard_logging_object_payload(
|
|||
cache_key=None,
|
||||
api_base=None,
|
||||
response_cost=None,
|
||||
litellm_overhead_time_ms=None,
|
||||
)
|
||||
)
|
||||
|
||||
|
|
|
|||
|
|
@ -17,4 +17,4 @@ general_settings:
|
|||
store_prompts_in_spend_logs: true
|
||||
|
||||
litellm_settings:
|
||||
callbacks: ["datadog_llm_observability"]
|
||||
callbacks: ["prometheus"]
|
||||
|
|
|
|||
|
|
@ -1490,6 +1490,7 @@ class StandardLoggingHiddenParams(TypedDict):
|
|||
cache_key: Optional[str]
|
||||
api_base: Optional[str]
|
||||
response_cost: Optional[str]
|
||||
litellm_overhead_time_ms: Optional[float]
|
||||
additional_headers: Optional[StandardLoggingAdditionalHeaders]
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -159,11 +159,6 @@ async def test_proxy_success_metrics():
|
|||
in metrics
|
||||
)
|
||||
|
||||
# assert (
|
||||
# 'litellm_deployment_latency_per_output_token_count{api_base="https://exampleopenaiendpoint-production.up.railway.app/",api_key_alias="None",api_provider="openai",hashed_api_key="88dc28d0f030c55ed4ab77ed8faf098196cb1c05df778539800c9f1243fe6b4b",litellm_model_name="fake",model_id="team-b-model",team="None",team_alias="None"}'
|
||||
# in metrics
|
||||
# )
|
||||
|
||||
verify_latency_metrics(metrics)
|
||||
|
||||
|
||||
|
|
@ -172,13 +167,19 @@ def verify_latency_metrics(metrics: str):
|
|||
Assert that LATENCY_BUCKETS distribution is used for
|
||||
- litellm_request_total_latency_metric_bucket
|
||||
- litellm_llm_api_latency_metric_bucket
|
||||
|
||||
Very important to verify that the overhead latency metric is present
|
||||
"""
|
||||
from litellm.types.integrations.prometheus import LATENCY_BUCKETS
|
||||
import re
|
||||
import time
|
||||
|
||||
time.sleep(2)
|
||||
|
||||
metric_names = [
|
||||
"litellm_request_total_latency_metric_bucket",
|
||||
"litellm_llm_api_latency_metric_bucket",
|
||||
"litellm_overhead_latency_metric_bucket",
|
||||
]
|
||||
|
||||
for metric_name in metric_names:
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue