diff --git a/litellm/llms/hosted_vllm/chat/transformation.py b/litellm/llms/hosted_vllm/chat/transformation.py index 1d21490ea31..e955800b947 100644 --- a/litellm/llms/hosted_vllm/chat/transformation.py +++ b/litellm/llms/hosted_vllm/chat/transformation.py @@ -23,7 +23,7 @@ from ...openai.chat.gpt_transformation import OpenAIGPTConfig class HostedVLLMChatConfig(OpenAIGPTConfig): def get_supported_openai_params(self, model: str) -> List[str]: params = super().get_supported_openai_params(model) - params.append("reasoning_effort") + params.extend(["reasoning_effort", "thinking"]) return params def map_openai_params( @@ -41,6 +41,27 @@ class HostedVLLMChatConfig(OpenAIGPTConfig): _tools = _remove_strict_from_schema(_tools) if _tools is not None: non_default_params["tools"] = _tools + + # Handle thinking parameter - convert Anthropic-style to OpenAI-style reasoning_effort + # vLLM is OpenAI-compatible, so it understands reasoning_effort, not thinking + # Reference: https://github.com/BerriAI/litellm/issues/19761 + thinking = non_default_params.pop("thinking", None) + if thinking is not None and isinstance(thinking, dict): + if thinking.get("type") == "enabled": + # Only convert if reasoning_effort not already set + if "reasoning_effort" not in non_default_params: + budget_tokens = thinking.get("budget_tokens", 0) + # Map budget_tokens to reasoning_effort level + # Same logic as Anthropic adapter (translate_anthropic_thinking_to_reasoning_effort) + if budget_tokens >= 10000: + non_default_params["reasoning_effort"] = "high" + elif budget_tokens >= 5000: + non_default_params["reasoning_effort"] = "medium" + elif budget_tokens >= 2000: + non_default_params["reasoning_effort"] = "low" + else: + non_default_params["reasoning_effort"] = "minimal" + return super().map_openai_params( non_default_params, optional_params, model, drop_params ) diff --git a/tests/test_litellm/llms/hosted_vllm/chat/test_hosted_vllm_chat_transformation.py b/tests/test_litellm/llms/hosted_vllm/chat/test_hosted_vllm_chat_transformation.py index 3749a5a8ca4..9b3b6aeaea1 100644 --- a/tests/test_litellm/llms/hosted_vllm/chat/test_hosted_vllm_chat_transformation.py +++ b/tests/test_litellm/llms/hosted_vllm/chat/test_hosted_vllm_chat_transformation.py @@ -101,3 +101,50 @@ def test_hosted_vllm_supports_reasoning_effort(): drop_params=False, ) assert optional_params["reasoning_effort"] == "high" + + +def test_hosted_vllm_supports_thinking(): + """ + Test that hosted_vllm supports the 'thinking' parameter. + + Anthropic-style thinking is converted to OpenAI-style reasoning_effort + since vLLM is OpenAI-compatible. + + Related issue: https://github.com/BerriAI/litellm/issues/19761 + """ + config = HostedVLLMChatConfig() + supported_params = config.get_supported_openai_params( + model="hosted_vllm/GLM-4.6-FP8" + ) + assert "thinking" in supported_params + + # Test thinking with low budget_tokens -> "minimal" (for < 2000) + optional_params = config.map_openai_params( + non_default_params={"thinking": {"type": "enabled", "budget_tokens": 1024}}, + optional_params={}, + model="hosted_vllm/GLM-4.6-FP8", + drop_params=False, + ) + assert "thinking" not in optional_params # thinking should NOT be passed + assert optional_params["reasoning_effort"] == "minimal" + + # Test thinking with high budget_tokens -> "high" + optional_params = config.map_openai_params( + non_default_params={"thinking": {"type": "enabled", "budget_tokens": 15000}}, + optional_params={}, + model="hosted_vllm/GLM-4.6-FP8", + drop_params=False, + ) + assert optional_params["reasoning_effort"] == "high" + + # Test that existing reasoning_effort is not overwritten + optional_params = config.map_openai_params( + non_default_params={ + "thinking": {"type": "enabled", "budget_tokens": 15000}, + "reasoning_effort": "low", + }, + optional_params={}, + model="hosted_vllm/GLM-4.6-FP8", + drop_params=False, + ) + assert optional_params["reasoning_effort"] == "low"