mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
fix(alerting): read hanging-request aliases from pre-call metadata location
This commit is contained in:
parent
24123269cc
commit
8cad7cb80b
3 changed files with 84 additions and 2 deletions
|
|
@ -14,7 +14,7 @@ from typing import TYPE_CHECKING, Any, Optional
|
|||
import litellm
|
||||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.caching.in_memory_cache import InMemoryCache
|
||||
from litellm.litellm_core_utils.core_helpers import get_litellm_metadata_from_kwargs
|
||||
from litellm.litellm_core_utils.core_helpers import get_request_metadata
|
||||
from litellm.types.integrations.slack_alerting import (
|
||||
HANGING_ALERT_BUFFER_TIME_SECONDS,
|
||||
MAX_OLDEST_HANGING_REQUESTS_TO_CHECK,
|
||||
|
|
@ -57,7 +57,7 @@ class AlertingHangingRequestCheck:
|
|||
if request_data is None:
|
||||
return
|
||||
|
||||
request_metadata = get_litellm_metadata_from_kwargs(kwargs=request_data)
|
||||
request_metadata = get_request_metadata(request_data=request_data)
|
||||
model = request_data.get("model", "")
|
||||
api_base: Optional[str] = None
|
||||
|
||||
|
|
|
|||
|
|
@ -234,6 +234,27 @@ def get_litellm_metadata_from_kwargs(kwargs: dict):
|
|||
return {}
|
||||
|
||||
|
||||
def get_request_metadata(request_data: dict) -> dict:
|
||||
"""
|
||||
Resolve proxy-internal metadata from request data at any point in the request lifecycle.
|
||||
|
||||
Proxy pre-call hooks see metadata at the top level of the request body; `litellm_params`
|
||||
only exists once the SDK completion call has built it, after which the same metadata is
|
||||
nested under it. Readers that can run on either side of that boundary must check both.
|
||||
"""
|
||||
nested_metadata = get_litellm_metadata_from_kwargs(kwargs=request_data)
|
||||
if nested_metadata:
|
||||
return nested_metadata
|
||||
|
||||
litellm_metadata = request_data.get("litellm_metadata")
|
||||
metadata = request_data.get("metadata")
|
||||
if isinstance(litellm_metadata, dict) and litellm_metadata:
|
||||
if isinstance(metadata, dict) and metadata:
|
||||
return add_missing_spend_metadata_to_litellm_metadata({**litellm_metadata}, metadata)
|
||||
return litellm_metadata
|
||||
return metadata if isinstance(metadata, dict) else {}
|
||||
|
||||
|
||||
def reconstruct_model_name(
|
||||
model_name: str,
|
||||
custom_llm_provider: Optional[str],
|
||||
|
|
|
|||
|
|
@ -81,6 +81,67 @@ class TestAlertingHangingRequestCheck:
|
|||
assert cached_data.request_id == "test_request_123"
|
||||
assert cached_data.model == "gpt-4"
|
||||
assert cached_data.api_base == "https://api.openai.com/v1"
|
||||
assert cached_data.key_alias == "test_key"
|
||||
assert cached_data.team_alias == "test_team"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_add_request_reads_aliases_from_top_level_litellm_metadata(
|
||||
self, hanging_request_checker
|
||||
):
|
||||
"""
|
||||
Batch/file routes keep proxy metadata in top-level `litellm_metadata`, which is
|
||||
also the only place it lives at pre-call time.
|
||||
"""
|
||||
request_data = {
|
||||
"litellm_call_id": "litellm_metadata_request",
|
||||
"model": "gpt-4",
|
||||
"litellm_metadata": {
|
||||
"user_api_key_alias": "batch_key",
|
||||
"user_api_key_team_alias": "batch_team",
|
||||
},
|
||||
}
|
||||
|
||||
await hanging_request_checker.add_request_to_hanging_request_check(request_data)
|
||||
|
||||
cached_data = (
|
||||
await hanging_request_checker.hanging_request_cache.async_get_cache(
|
||||
key="litellm_metadata_request"
|
||||
)
|
||||
)
|
||||
|
||||
assert cached_data.key_alias == "batch_key"
|
||||
assert cached_data.team_alias == "batch_team"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_add_request_prefers_litellm_params_metadata_when_present(
|
||||
self, hanging_request_checker
|
||||
):
|
||||
"""
|
||||
Post-call kwargs carry metadata under `litellm_params`; that copy wins over a
|
||||
stale top-level one.
|
||||
"""
|
||||
request_data = {
|
||||
"litellm_call_id": "litellm_params_request",
|
||||
"model": "gpt-4",
|
||||
"metadata": {"user_api_key_alias": "stale_key"},
|
||||
"litellm_params": {
|
||||
"metadata": {
|
||||
"user_api_key_alias": "nested_key",
|
||||
"user_api_key_team_alias": "nested_team",
|
||||
}
|
||||
},
|
||||
}
|
||||
|
||||
await hanging_request_checker.add_request_to_hanging_request_check(request_data)
|
||||
|
||||
cached_data = (
|
||||
await hanging_request_checker.hanging_request_cache.async_get_cache(
|
||||
key="litellm_params_request"
|
||||
)
|
||||
)
|
||||
|
||||
assert cached_data.key_alias == "nested_key"
|
||||
assert cached_data.team_alias == "nested_team"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_add_request_to_hanging_request_check_none_request_data(
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue