[Feat] Add reasoning_effort param for hosted_vllm provider (#13620)

* add reasoning_effort to hosted_vllm

* test_hosted_vllm_supports_reasoning_effort

* Reasoning Effort
This commit is contained in:
Ishaan Jaff 2025-08-14 10:10:30 -07:00 • committed by GitHub
parent dea98a315b
commit 5bb96af818
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
3 changed files with 66 additions and 0 deletions

View file

@ -104,6 +104,52 @@ Here's how to call an OpenAI-Compatible Endpoint with the LiteLLM Proxy Server
</Tabs>
## Reasoning Effort
<Tabs>
<TabItem value="sdk" label="SDK">
```python
from litellm import completion
response = completion(
model="hosted_vllm/gpt-oss-120b",
messages=[{"role": "user", "content": "whats 2 + 2"}],
reasoning_effort="high",
api_base="https://hosted-vllm-api.co",
)
print(response)
```
</TabItem>
<TabItem value="proxy" label="PROXY">
1. Setup config.yaml
```yaml
model_list:
- model_name: gpt-oss-120b
litellm_params:
model: hosted_vllm/gpt-oss-120b
api_base: https://hosted-vllm-api.co
```
2. Start the proxy
```bash
litellm --config /path/to/config.yaml
```
3. Test it!
```bash
curl http://0.0.0.0:4000/v1/chat/completions \
-H "Content-Type: application/json" \
-d '{"model": "gpt-oss-120b", "messages": [{"role": "user", "content": "whats 2 + 2"}], "reasoning_effort": "high"}'
```
</TabItem>
</Tabs>
## Embeddings

View file

@ -21,6 +21,11 @@ from ...openai.chat.gpt_transformation import OpenAIGPTConfig
class HostedVLLMChatConfig(OpenAIGPTConfig):
def get_supported_openai_params(self, model: str) -> List[str]:
params = super().get_supported_openai_params(model)
params.append("reasoning_effort")
return params
def map_openai_params(
self,
non_default_params: dict,

View file

@ -86,3 +86,18 @@ def test_hosted_vllm_chat_transformation_with_audio_url():
],
}
]
def test_hosted_vllm_supports_reasoning_effort():
config = HostedVLLMChatConfig()
supported_params = config.get_supported_openai_params(
model="hosted_vllm/gpt-oss-120b"
)
assert "reasoning_effort" in supported_params
optional_params = config.map_openai_params(
non_default_params={"reasoning_effort": "high"},
optional_params={},
model="hosted_vllm/gpt-oss-120b",
drop_params=False,
)
assert optional_params["reasoning_effort"] == "high"