diff --git a/litellm/proxy/health_check.py b/litellm/proxy/health_check.py index 7c340ff5df6..400da9da0d5 100644 --- a/litellm/proxy/health_check.py +++ b/litellm/proxy/health_check.py @@ -36,6 +36,11 @@ ADMIN_ONLY_HEALTH_DISPLAY_PARAMS = ("api_base", "api_version") MINIMAL_DISPLAY_PARAMS = ["model", "mode_error"] +# Health-check modes that forward `reasoning_effort` to the provider (chat-style calls). +_HEALTH_CHECK_MODES_SUPPORTING_REASONING_EFFORT = frozenset( + (None, "chat", "completion") +) + def _get_process_rss_mb() -> Optional[float]: """ @@ -375,6 +380,12 @@ def _update_litellm_params_for_health_check( if _resolved_max_tokens is not None: litellm_params["max_tokens"] = _resolved_max_tokens + # Per-model reasoning effort for health checks only (e.g. reasoning_effort=none). + if model_info.get("mode", None) in _HEALTH_CHECK_MODES_SUPPORTING_REASONING_EFFORT: + _hc_reasoning_effort = model_info.get("health_check_reasoning_effort", None) + if _hc_reasoning_effort is not None: + litellm_params["reasoning_effort"] = _hc_reasoning_effort + _health_check_model = model_info.get("health_check_model", None) if _health_check_model is not None: litellm_params["model"] = _health_check_model diff --git a/tests/test_litellm/proxy/test_health_check_max_tokens.py b/tests/test_litellm/proxy/test_health_check_max_tokens.py index 09211b72c3e..5292d2af061 100644 --- a/tests/test_litellm/proxy/test_health_check_max_tokens.py +++ b/tests/test_litellm/proxy/test_health_check_max_tokens.py @@ -225,3 +225,37 @@ def test_wildcard_ignores_reasoning_split_model_info(monkeypatch): litellm_params = {"model": "openai/*"} assert _resolve_health_check_max_tokens(model_info, litellm_params) is None + + +def test_update_litellm_params_health_check_reasoning_effort(): + """model_info.health_check_reasoning_effort sets reasoning_effort for chat-style health checks.""" + model_info = {"health_check_reasoning_effort": "low"} + litellm_params = {"model": "openai/gpt-5", "api_key": "x"} + out = _update_litellm_params_for_health_check(model_info, dict(litellm_params)) + assert out.get("reasoning_effort") == "low" + + model_info = {"mode": "chat", "health_check_reasoning_effort": "none"} + out = _update_litellm_params_for_health_check( + model_info, {"model": "openai/gpt-5", "api_key": "x"} + ) + assert out.get("reasoning_effort") == "none" + + model_info = { + "health_check_reasoning_effort": {"effort": "none", "summary": "auto"}, + } + out = _update_litellm_params_for_health_check( + model_info, {"model": "openai/gpt-5.1", "api_key": "x"} + ) + assert out.get("reasoning_effort") == {"effort": "none", "summary": "auto"} + + model_info = {"mode": "embedding", "health_check_reasoning_effort": "low"} + out = _update_litellm_params_for_health_check( + model_info, {"model": "text-embedding-3-small", "api_key": "x"} + ) + assert "reasoning_effort" not in out + + model_info = {} + out = _update_litellm_params_for_health_check( + model_info, {"model": "openai/gpt-4o", "api_key": "x"} + ) + assert "reasoning_effort" not in out