diff --git a/litellm/litellm_core_utils/health_check_helpers.py b/litellm/litellm_core_utils/health_check_helpers.py index 626705c7267..2ba5661db15 100644 --- a/litellm/litellm_core_utils/health_check_helpers.py +++ b/litellm/litellm_core_utils/health_check_helpers.py @@ -157,9 +157,7 @@ class HealthCheckHelpers: **model_params, ), "completion": lambda: litellm.atext_completion( - **_filter_model_params( - model_params=model_params, keep_max_tokens=True - ), + **_filter_model_params(model_params=model_params, keep_max_tokens=True), prompt=prompt or "test", ), "embedding": lambda: litellm.aembedding( diff --git a/litellm/litellm_core_utils/health_check_utils.py b/litellm/litellm_core_utils/health_check_utils.py index b0e71896728..6622e901d0d 100644 --- a/litellm/litellm_core_utils/health_check_utils.py +++ b/litellm/litellm_core_utils/health_check_utils.py @@ -13,14 +13,10 @@ _COMPLETION_HEALTH_CHECK_STRIP_KEYS = {"messages"} #: ``400 Unknown parameter: 'max_tokens'``. ``atext_completion`` does accept #: ``max_tokens``, so the ``completion`` mode keeps it (preserves the #: ``BACKGROUND_HEALTH_CHECK_MAX_TOKENS`` cost cap). -_NON_CHAT_HEALTH_CHECK_STRIP_KEYS = _COMPLETION_HEALTH_CHECK_STRIP_KEYS | { - "max_tokens" -} +_NON_CHAT_HEALTH_CHECK_STRIP_KEYS = _COMPLETION_HEALTH_CHECK_STRIP_KEYS | {"max_tokens"} -def _filter_model_params( - model_params: dict, *, keep_max_tokens: bool = False -) -> dict: +def _filter_model_params(model_params: dict, *, keep_max_tokens: bool = False) -> dict: """Strip chat-only params before invoking a non-chat health check handler. ``litellm.acompletion`` is the only mode handler that consumes diff --git a/tests/litellm/litellm_core_utils/test_health_check_utils.py b/tests/litellm/litellm_core_utils/test_health_check_utils.py index 944f4908523..ccb4a72239a 100644 --- a/tests/litellm/litellm_core_utils/test_health_check_utils.py +++ b/tests/litellm/litellm_core_utils/test_health_check_utils.py @@ -43,9 +43,7 @@ def test_filter_keeps_max_tokens_for_completion_mode(): ``litellm.atext_completion`` accepts ``max_tokens``; stripping it would silently remove the ``BACKGROUND_HEALTH_CHECK_MAX_TOKENS`` cost cap. """ - filtered = _filter_model_params( - model_params=_sample_params(), keep_max_tokens=True - ) + filtered = _filter_model_params(model_params=_sample_params(), keep_max_tokens=True) assert "messages" not in filtered assert filtered.get("max_tokens") == 5