diff --git a/litellm/litellm_core_utils/health_check_helpers.py b/litellm/litellm_core_utils/health_check_helpers.py index a0f027cd58f..d1381e06bcf 100644 --- a/litellm/litellm_core_utils/health_check_helpers.py +++ b/litellm/litellm_core_utils/health_check_helpers.py @@ -251,7 +251,7 @@ class HealthCheckHelpers: ), "responses": lambda: litellm.aresponses( **_filter_model_params(model_params=model_params), - input=prompt or "test", + input=[{"role": "user", "content": prompt or "test"}], ), "ocr": lambda: litellm.aocr( **_filter_model_params(model_params=model_params), diff --git a/litellm/main.py b/litellm/main.py index 6c85adf3ae8..3923dc91701 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -8755,6 +8755,10 @@ async def ahealth_check( mode = litellm.model_cost[model].get("mode") model_params["cache"] = {"no-cache": True} # don't used cached responses for making health check calls + # chatgpt provider uses the Responses API exclusively; default to "responses" so the + # health check doesn't send a Chat Completions payload that the backend rejects. + if mode is None and custom_llm_provider == "chatgpt": + mode = "responses" mode = mode or "chat" if "*" in model: return await HealthCheckHelpers.ahealth_check_wildcard_models(