feat(hosted_vllm): support thinking parameter in anthropic_messages() and .completion()

feat(hosted_vllm): support `thinking` parameter in `anthropic_messages()` and `.completion()`
This commit is contained in:
mubashir1osmani 2026-01-27 22:13:53 -05:00 • committed by GitHub
commit 9a245031bd
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
2 changed files with 69 additions and 1 deletions

View file

@ -23,7 +23,7 @@ from ...openai.chat.gpt_transformation import OpenAIGPTConfig
class HostedVLLMChatConfig(OpenAIGPTConfig):
def get_supported_openai_params(self, model: str) -> List[str]:
params = super().get_supported_openai_params(model)
params.append("reasoning_effort")
params.extend(["reasoning_effort", "thinking"])
return params
def map_openai_params(
@ -41,6 +41,27 @@ class HostedVLLMChatConfig(OpenAIGPTConfig):
_tools = _remove_strict_from_schema(_tools)
if _tools is not None:
non_default_params["tools"] = _tools
# Handle thinking parameter - convert Anthropic-style to OpenAI-style reasoning_effort
# vLLM is OpenAI-compatible, so it understands reasoning_effort, not thinking
# Reference: https://github.com/BerriAI/litellm/issues/19761
thinking = non_default_params.pop("thinking", None)
if thinking is not None and isinstance(thinking, dict):
if thinking.get("type") == "enabled":
# Only convert if reasoning_effort not already set
if "reasoning_effort" not in non_default_params:
budget_tokens = thinking.get("budget_tokens", 0)
# Map budget_tokens to reasoning_effort level
# Same logic as Anthropic adapter (translate_anthropic_thinking_to_reasoning_effort)
if budget_tokens >= 10000:
non_default_params["reasoning_effort"] = "high"
elif budget_tokens >= 5000:
non_default_params["reasoning_effort"] = "medium"
elif budget_tokens >= 2000:
non_default_params["reasoning_effort"] = "low"
else:
non_default_params["reasoning_effort"] = "minimal"
return super().map_openai_params(
non_default_params, optional_params, model, drop_params
)

View file

@ -101,3 +101,50 @@ def test_hosted_vllm_supports_reasoning_effort():
drop_params=False,
)
assert optional_params["reasoning_effort"] == "high"
def test_hosted_vllm_supports_thinking():
"""
Test that hosted_vllm supports the 'thinking' parameter.
Anthropic-style thinking is converted to OpenAI-style reasoning_effort
since vLLM is OpenAI-compatible.
Related issue: https://github.com/BerriAI/litellm/issues/19761
"""
config = HostedVLLMChatConfig()
supported_params = config.get_supported_openai_params(
model="hosted_vllm/GLM-4.6-FP8"
)
assert "thinking" in supported_params
# Test thinking with low budget_tokens -> "minimal" (for < 2000)
optional_params = config.map_openai_params(
non_default_params={"thinking": {"type": "enabled", "budget_tokens": 1024}},
optional_params={},
model="hosted_vllm/GLM-4.6-FP8",
drop_params=False,
)
assert "thinking" not in optional_params # thinking should NOT be passed
assert optional_params["reasoning_effort"] == "minimal"
# Test thinking with high budget_tokens -> "high"
optional_params = config.map_openai_params(
non_default_params={"thinking": {"type": "enabled", "budget_tokens": 15000}},
optional_params={},
model="hosted_vllm/GLM-4.6-FP8",
drop_params=False,
)
assert optional_params["reasoning_effort"] == "high"
# Test that existing reasoning_effort is not overwritten
optional_params = config.map_openai_params(
non_default_params={
"thinking": {"type": "enabled", "budget_tokens": 15000},
"reasoning_effort": "low",
},
optional_params={},
model="hosted_vllm/GLM-4.6-FP8",
drop_params=False,
)
assert optional_params["reasoning_effort"] == "low"