mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
Merge pull request #39190 from WolframRavenwolf/litellm_wandb_reasoning_effort
fix(wandb): preserve reasoning_effort in chat completions
This commit is contained in:
commit
2f114d44ed
4 changed files with 239 additions and 1 deletions
|
|
@ -6,10 +6,17 @@ This is OpenAI compatible - no translation needed / occurs
|
|||
|
||||
from typing import Final
|
||||
|
||||
import litellm
|
||||
from litellm.llms.openai.chat.gpt_transformation import OpenAIGPTConfig
|
||||
|
||||
|
||||
class WandbConfig(OpenAIGPTConfig):
|
||||
def get_supported_openai_params(self, model: str) -> list[str]: # mutable-ok: inherited contract
|
||||
supported_params: Final = super().get_supported_openai_params(model)
|
||||
if litellm.supports_reasoning(model=model, custom_llm_provider="wandb"):
|
||||
return supported_params + ["reasoning_effort"] # mutable-ok: inherited contract
|
||||
return supported_params
|
||||
|
||||
def map_openai_params(
|
||||
self,
|
||||
non_default_params: dict,
|
||||
|
|
|
|||
|
|
@ -48655,6 +48655,7 @@
|
|||
"output_cost_per_token": 0.0
|
||||
},
|
||||
"wandb/openai/gpt-oss-120b": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 131072,
|
||||
"max_output_tokens": 131072,
|
||||
|
|
@ -48665,6 +48666,7 @@
|
|||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/openai/gpt-oss-20b": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 131072,
|
||||
"max_output_tokens": 131072,
|
||||
|
|
@ -48675,6 +48677,7 @@
|
|||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/zai-org/GLM-4.5": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 131072,
|
||||
"max_output_tokens": 131072,
|
||||
|
|
@ -48703,6 +48706,7 @@
|
|||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/Qwen/Qwen3-235B-A22B-Thinking-2507": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
|
|
@ -48759,6 +48763,7 @@
|
|||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/deepseek-ai/DeepSeek-V3.1": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 128000,
|
||||
"max_input_tokens": 161000,
|
||||
"max_output_tokens": 128000,
|
||||
|
|
@ -48769,6 +48774,7 @@
|
|||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/deepseek-ai/DeepSeek-R1-0528": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 161000,
|
||||
"max_input_tokens": 161000,
|
||||
"max_output_tokens": 161000,
|
||||
|
|
@ -58630,6 +58636,7 @@
|
|||
"supports_vision": false
|
||||
},
|
||||
"wandb/deepseek-ai/DeepSeek-V4-Flash": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 1048576,
|
||||
"max_input_tokens": 1048576,
|
||||
"input_cost_per_token": 1.4e-07,
|
||||
|
|
@ -58642,6 +58649,7 @@
|
|||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/deepseek-ai/DeepSeek-V4-Flash-0731": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 262144,
|
||||
"input_cost_per_token": 1.3e-07,
|
||||
|
|
@ -58654,6 +58662,7 @@
|
|||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/deepseek-ai/DeepSeek-V4-Pro": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 1048576,
|
||||
"max_input_tokens": 1048576,
|
||||
"input_cost_per_token": 1.15e-06,
|
||||
|
|
@ -58666,6 +58675,7 @@
|
|||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/google/gemma-4-31B-it": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 262144,
|
||||
"input_cost_per_token": 1e-07,
|
||||
|
|
@ -58706,6 +58716,7 @@
|
|||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/MiniMaxAI/MiniMax-M3": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 262144,
|
||||
"input_cost_per_token": 2.3e-07,
|
||||
|
|
@ -58718,6 +58729,7 @@
|
|||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/moonshotai/Kimi-K2.7-Code": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 262144,
|
||||
"input_cost_per_token": 7.1e-07,
|
||||
|
|
@ -58730,6 +58742,7 @@
|
|||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/moonshotai/Kimi-K2.6": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 262144,
|
||||
"input_cost_per_token": 6.5e-07,
|
||||
|
|
@ -58742,6 +58755,7 @@
|
|||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 262144,
|
||||
"input_cost_per_token": 1e-07,
|
||||
|
|
@ -58754,6 +58768,7 @@
|
|||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 262144,
|
||||
"input_cost_per_token": 7.5e-07,
|
||||
|
|
@ -58776,6 +58791,7 @@
|
|||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/Qwen/Qwen3.8-27B": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 262144,
|
||||
"input_cost_per_token": 4e-07,
|
||||
|
|
@ -58788,6 +58804,7 @@
|
|||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/Qwen/Qwen3.6-35B-A3B": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 262144,
|
||||
"input_cost_per_token": 2.5e-07,
|
||||
|
|
@ -58798,6 +58815,7 @@
|
|||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/Qwen/Qwen3.6-27B": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 262144,
|
||||
"input_cost_per_token": 6e-07,
|
||||
|
|
@ -58810,6 +58828,7 @@
|
|||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/Qwen/Qwen3.5-35B-A3B": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 262144,
|
||||
"input_cost_per_token": 2.5e-07,
|
||||
|
|
@ -58829,7 +58848,28 @@
|
|||
"supports_vision": false,
|
||||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/deepseek-ai/DeepSeek-V4-Pro-0813": {
|
||||
"litellm_provider": "wandb",
|
||||
"mode": "chat",
|
||||
"supports_reasoning": true,
|
||||
"input_cost_per_token": 0.00000131,
|
||||
"output_cost_per_token": 0.00000396,
|
||||
"cache_read_input_token_cost": 0.000000044,
|
||||
"supports_prompt_caching": true,
|
||||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/ibm-granite/granite-4.2-8b": {
|
||||
"litellm_provider": "wandb",
|
||||
"mode": "chat",
|
||||
"supports_reasoning": true,
|
||||
"input_cost_per_token": 0.0000001,
|
||||
"output_cost_per_token": 0.00000015,
|
||||
"cache_read_input_token_cost": 0.00000005,
|
||||
"supports_prompt_caching": true,
|
||||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/zai-org/GLM-5.2": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 262144,
|
||||
"input_cost_per_token": 7.6e-07,
|
||||
|
|
|
|||
|
|
@ -48655,6 +48655,7 @@
|
|||
"output_cost_per_token": 0.0
|
||||
},
|
||||
"wandb/openai/gpt-oss-120b": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 131072,
|
||||
"max_output_tokens": 131072,
|
||||
|
|
@ -48665,6 +48666,7 @@
|
|||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/openai/gpt-oss-20b": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 131072,
|
||||
"max_output_tokens": 131072,
|
||||
|
|
@ -48675,6 +48677,7 @@
|
|||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/zai-org/GLM-4.5": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 131072,
|
||||
"max_output_tokens": 131072,
|
||||
|
|
@ -48703,6 +48706,7 @@
|
|||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/Qwen/Qwen3-235B-A22B-Thinking-2507": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
|
|
@ -48759,6 +48763,7 @@
|
|||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/deepseek-ai/DeepSeek-V3.1": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 128000,
|
||||
"max_input_tokens": 161000,
|
||||
"max_output_tokens": 128000,
|
||||
|
|
@ -48769,6 +48774,7 @@
|
|||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/deepseek-ai/DeepSeek-R1-0528": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 161000,
|
||||
"max_input_tokens": 161000,
|
||||
"max_output_tokens": 161000,
|
||||
|
|
@ -58630,6 +58636,7 @@
|
|||
"supports_vision": false
|
||||
},
|
||||
"wandb/deepseek-ai/DeepSeek-V4-Flash": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 1048576,
|
||||
"max_input_tokens": 1048576,
|
||||
"input_cost_per_token": 1.4e-07,
|
||||
|
|
@ -58642,6 +58649,7 @@
|
|||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/deepseek-ai/DeepSeek-V4-Flash-0731": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 262144,
|
||||
"input_cost_per_token": 1.3e-07,
|
||||
|
|
@ -58654,6 +58662,7 @@
|
|||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/deepseek-ai/DeepSeek-V4-Pro": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 1048576,
|
||||
"max_input_tokens": 1048576,
|
||||
"input_cost_per_token": 1.15e-06,
|
||||
|
|
@ -58666,6 +58675,7 @@
|
|||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/google/gemma-4-31B-it": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 262144,
|
||||
"input_cost_per_token": 1e-07,
|
||||
|
|
@ -58706,6 +58716,7 @@
|
|||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/MiniMaxAI/MiniMax-M3": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 262144,
|
||||
"input_cost_per_token": 2.3e-07,
|
||||
|
|
@ -58718,6 +58729,7 @@
|
|||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/moonshotai/Kimi-K2.7-Code": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 262144,
|
||||
"input_cost_per_token": 7.1e-07,
|
||||
|
|
@ -58730,6 +58742,7 @@
|
|||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/moonshotai/Kimi-K2.6": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 262144,
|
||||
"input_cost_per_token": 6.5e-07,
|
||||
|
|
@ -58742,6 +58755,7 @@
|
|||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 262144,
|
||||
"input_cost_per_token": 1e-07,
|
||||
|
|
@ -58754,6 +58768,7 @@
|
|||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 262144,
|
||||
"input_cost_per_token": 7.5e-07,
|
||||
|
|
@ -58776,6 +58791,7 @@
|
|||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/Qwen/Qwen3.8-27B": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 262144,
|
||||
"input_cost_per_token": 4e-07,
|
||||
|
|
@ -58788,6 +58804,7 @@
|
|||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/Qwen/Qwen3.6-35B-A3B": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 262144,
|
||||
"input_cost_per_token": 2.5e-07,
|
||||
|
|
@ -58798,6 +58815,7 @@
|
|||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/Qwen/Qwen3.6-27B": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 262144,
|
||||
"input_cost_per_token": 6e-07,
|
||||
|
|
@ -58810,6 +58828,7 @@
|
|||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/Qwen/Qwen3.5-35B-A3B": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 262144,
|
||||
"input_cost_per_token": 2.5e-07,
|
||||
|
|
@ -58829,7 +58848,28 @@
|
|||
"supports_vision": false,
|
||||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/deepseek-ai/DeepSeek-V4-Pro-0813": {
|
||||
"litellm_provider": "wandb",
|
||||
"mode": "chat",
|
||||
"supports_reasoning": true,
|
||||
"input_cost_per_token": 0.00000131,
|
||||
"output_cost_per_token": 0.00000396,
|
||||
"cache_read_input_token_cost": 0.000000044,
|
||||
"supports_prompt_caching": true,
|
||||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/ibm-granite/granite-4.2-8b": {
|
||||
"litellm_provider": "wandb",
|
||||
"mode": "chat",
|
||||
"supports_reasoning": true,
|
||||
"input_cost_per_token": 0.0000001,
|
||||
"output_cost_per_token": 0.00000015,
|
||||
"cache_read_input_token_cost": 0.00000005,
|
||||
"supports_prompt_caching": true,
|
||||
"source": "https://wandb.ai/site/pricing/tokens/"
|
||||
},
|
||||
"wandb/zai-org/GLM-5.2": {
|
||||
"supports_reasoning": true,
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 262144,
|
||||
"input_cost_per_token": 7.6e-07,
|
||||
|
|
|
|||
|
|
@ -5,18 +5,92 @@ These tests validate the WandbInferenceConfig class which extends OpenAIGPTConfi
|
|||
Nebius AI Studio is an OpenAI-compatible provider with minor customizations.
|
||||
"""
|
||||
|
||||
|
||||
import json
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
import respx
|
||||
|
||||
import litellm
|
||||
from litellm import completion
|
||||
from litellm.llms.wandb.chat.transformation import WandbConfig
|
||||
|
||||
|
||||
WANDB_REASONING_MODELS: Final = (
|
||||
"deepseek-ai/DeepSeek-V4-Flash",
|
||||
"deepseek-ai/DeepSeek-V4-Flash-0731",
|
||||
"deepseek-ai/DeepSeek-V4-Pro",
|
||||
"deepseek-ai/DeepSeek-V4-Pro-0813",
|
||||
"deepseek-ai/DeepSeek-V3.1",
|
||||
"google/gemma-4-31B-it",
|
||||
"ibm-granite/granite-4.2-8b",
|
||||
"MiniMaxAI/MiniMax-M3",
|
||||
"moonshotai/Kimi-K2.7-Code",
|
||||
"moonshotai/Kimi-K2.6",
|
||||
"nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B",
|
||||
"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B",
|
||||
"openai/gpt-oss-120b",
|
||||
"openai/gpt-oss-20b",
|
||||
"Qwen/Qwen3.8-27B",
|
||||
"Qwen/Qwen3.6-35B-A3B",
|
||||
"Qwen/Qwen3.6-27B",
|
||||
"Qwen/Qwen3.5-35B-A3B",
|
||||
"zai-org/GLM-5.2",
|
||||
"moonshotai/Kimi-K2.5",
|
||||
"MiniMaxAI/MiniMax-M2.5",
|
||||
"zai-org/GLM-4.5",
|
||||
"Qwen/Qwen3-235B-A22B-Thinking-2507",
|
||||
"deepseek-ai/DeepSeek-R1-0528",
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def wandb_test_config(local_model_cost_map, monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setattr(litellm, "disable_aiohttp_transport", True)
|
||||
monkeypatch.setattr(litellm, "telemetry", False)
|
||||
monkeypatch.setattr(litellm, "drop_params", False)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def wandb_request_mock(respx_mock: respx.MockRouter) -> respx.Route:
|
||||
return respx_mock.post("https://api.inference.wandb.ai/v1/chat/completions").respond(
|
||||
json={
|
||||
"id": "chatcmpl-123",
|
||||
"object": "chat.completion",
|
||||
"created": 1677652288,
|
||||
"model": "test-model",
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"message": {"role": "assistant", "content": "Done"},
|
||||
"finish_reason": "stop",
|
||||
}
|
||||
],
|
||||
"usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2},
|
||||
},
|
||||
status_code=200,
|
||||
)
|
||||
|
||||
|
||||
class TestWandbConfig:
|
||||
"""Test class for WandB Inference functionality"""
|
||||
|
||||
@pytest.mark.parametrize("model", WANDB_REASONING_MODELS)
|
||||
def test_map_openai_params_preserves_reasoning_effort(self, wandb_test_config, model: str):
|
||||
assert litellm.model_cost[f"wandb/{model}"].get("supports_reasoning") is True
|
||||
supported_params = litellm.get_supported_openai_params(model=f"wandb/{model}")
|
||||
assert supported_params is not None
|
||||
assert "reasoning_effort" in supported_params
|
||||
|
||||
result = WandbConfig().map_openai_params(
|
||||
non_default_params={"reasoning_effort": "medium", "max_completion_tokens": 64},
|
||||
optional_params={},
|
||||
model=model,
|
||||
drop_params=True,
|
||||
)
|
||||
|
||||
assert result == {"reasoning_effort": "medium", "max_tokens": 64}
|
||||
|
||||
def test_default_api_base(self):
|
||||
"""Test that default API base is used when none is provided"""
|
||||
config = WandbConfig()
|
||||
|
|
@ -139,3 +213,80 @@ class TestWandbConfig:
|
|||
# Check for specific content in the response
|
||||
assert "```python" in content
|
||||
assert "Hey from LiteLLM" in content
|
||||
|
||||
@pytest.mark.respx()
|
||||
@pytest.mark.parametrize(
|
||||
"model,effort",
|
||||
tuple((model, "medium") for model in WANDB_REASONING_MODELS)
|
||||
+ (
|
||||
("Qwen/Qwen3.8-27B", "low"),
|
||||
("Qwen/Qwen3.8-27B", "xhigh"),
|
||||
),
|
||||
)
|
||||
def test_wandb_completion_preserves_reasoning_effort_with_drop_params(
|
||||
self, wandb_test_config, wandb_request_mock: respx.Route, model: str, effort: str
|
||||
):
|
||||
completion(
|
||||
model=f"wandb/{model}",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
api_key="fake-wandb-key",
|
||||
api_base="https://api.inference.wandb.ai/v1",
|
||||
reasoning_effort=effort,
|
||||
max_completion_tokens=64,
|
||||
drop_params=True,
|
||||
)
|
||||
|
||||
assert wandb_request_mock.call_count == 1
|
||||
request_body = json.loads(wandb_request_mock.calls[0].request.content)
|
||||
assert request_body["model"] == model
|
||||
assert request_body["reasoning_effort"] == effort
|
||||
assert request_body["max_tokens"] == 64
|
||||
assert "max_completion_tokens" not in request_body
|
||||
|
||||
@pytest.mark.respx(assert_all_called=False)
|
||||
@pytest.mark.parametrize("drop_params", [True, False])
|
||||
@pytest.mark.parametrize(
|
||||
"model,explicit_false",
|
||||
[
|
||||
("meta-llama/Llama-3.1-8B-Instruct", False),
|
||||
("unknown-model", False),
|
||||
("openai/gpt-oss-20b", True),
|
||||
],
|
||||
)
|
||||
def test_wandb_completion_without_reasoning_support(
|
||||
self,
|
||||
wandb_test_config,
|
||||
wandb_request_mock: respx.Route,
|
||||
respx_mock: respx.MockRouter,
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
model: str,
|
||||
explicit_false: bool,
|
||||
drop_params: bool,
|
||||
):
|
||||
with monkeypatch.context() as context:
|
||||
if explicit_false:
|
||||
context.setitem(litellm.model_cost[f"wandb/{model}"], "supports_reasoning", False)
|
||||
|
||||
kwargs = {
|
||||
"model": f"wandb/{model}",
|
||||
"messages": [{"role": "user", "content": "Hello"}],
|
||||
"api_key": "fake-wandb-key",
|
||||
"api_base": "https://api.inference.wandb.ai/v1",
|
||||
"reasoning_effort": "medium",
|
||||
"drop_params": drop_params,
|
||||
}
|
||||
if not drop_params:
|
||||
with pytest.raises(litellm.UnsupportedParamsError, match="reasoning_effort"):
|
||||
completion(**kwargs)
|
||||
assert len(respx_mock.calls) == 0
|
||||
return
|
||||
|
||||
completion(**kwargs)
|
||||
assert wandb_request_mock.call_count == 1
|
||||
request_body = json.loads(wandb_request_mock.calls[0].request.content)
|
||||
assert request_body["model"] == model
|
||||
assert "reasoning_effort" not in request_body
|
||||
|
||||
supported_params = litellm.get_supported_openai_params(model=f"wandb/{model}")
|
||||
assert supported_params is not None
|
||||
assert "reasoning_effort" not in supported_params
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue