mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-11 03:38:38 +00:00
[Feat] Add reasoning_effort param for hosted_vllm provider (#13620)
* add reasoning_effort to hosted_vllm * test_hosted_vllm_supports_reasoning_effort * Reasoning Effort
This commit is contained in:
parent
dea98a315b
commit
5bb96af818
3 changed files with 66 additions and 0 deletions
|
|
@ -104,6 +104,52 @@ Here's how to call an OpenAI-Compatible Endpoint with the LiteLLM Proxy Server
|
|||
|
||||
</Tabs>
|
||||
|
||||
## Reasoning Effort
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
|
||||
response = completion(
|
||||
model="hosted_vllm/gpt-oss-120b",
|
||||
messages=[{"role": "user", "content": "whats 2 + 2"}],
|
||||
reasoning_effort="high",
|
||||
api_base="https://hosted-vllm-api.co",
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
1. Setup config.yaml
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-oss-120b
|
||||
litellm_params:
|
||||
model: hosted_vllm/gpt-oss-120b
|
||||
api_base: https://hosted-vllm-api.co
|
||||
```
|
||||
|
||||
2. Start the proxy
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
3. Test it!
|
||||
|
||||
```bash
|
||||
curl http://0.0.0.0:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{"model": "gpt-oss-120b", "messages": [{"role": "user", "content": "whats 2 + 2"}], "reasoning_effort": "high"}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
|
||||
## Embeddings
|
||||
|
||||
|
|
|
|||
|
|
@ -21,6 +21,11 @@ from ...openai.chat.gpt_transformation import OpenAIGPTConfig
|
|||
|
||||
|
||||
class HostedVLLMChatConfig(OpenAIGPTConfig):
|
||||
def get_supported_openai_params(self, model: str) -> List[str]:
|
||||
params = super().get_supported_openai_params(model)
|
||||
params.append("reasoning_effort")
|
||||
return params
|
||||
|
||||
def map_openai_params(
|
||||
self,
|
||||
non_default_params: dict,
|
||||
|
|
|
|||
|
|
@ -86,3 +86,18 @@ def test_hosted_vllm_chat_transformation_with_audio_url():
|
|||
],
|
||||
}
|
||||
]
|
||||
|
||||
|
||||
def test_hosted_vllm_supports_reasoning_effort():
|
||||
config = HostedVLLMChatConfig()
|
||||
supported_params = config.get_supported_openai_params(
|
||||
model="hosted_vllm/gpt-oss-120b"
|
||||
)
|
||||
assert "reasoning_effort" in supported_params
|
||||
optional_params = config.map_openai_params(
|
||||
non_default_params={"reasoning_effort": "high"},
|
||||
optional_params={},
|
||||
model="hosted_vllm/gpt-oss-120b",
|
||||
drop_params=False,
|
||||
)
|
||||
assert optional_params["reasoning_effort"] == "high"
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue