Merge pull request #39190 from WolframRavenwolf/litellm_wandb_reasoning_effort

fix(wandb): preserve reasoning_effort in chat completions
This commit is contained in:
ryan-crabbe-berri 2026-09-10 14:02:49 -07:00 • committed by GitHub
commit 2f114d44ed
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
4 changed files with 239 additions and 1 deletions

View file

@ -6,10 +6,17 @@ This is OpenAI compatible - no translation needed / occurs
from typing import Final
import litellm
from litellm.llms.openai.chat.gpt_transformation import OpenAIGPTConfig
class WandbConfig(OpenAIGPTConfig):
def get_supported_openai_params(self, model: str) -> list[str]: # mutable-ok: inherited contract
supported_params: Final = super().get_supported_openai_params(model)
if litellm.supports_reasoning(model=model, custom_llm_provider="wandb"):
return supported_params + ["reasoning_effort"] # mutable-ok: inherited contract
return supported_params
def map_openai_params(
self,
non_default_params: dict,

View file

@ -48655,6 +48655,7 @@
"output_cost_per_token": 0.0
},
"wandb/openai/gpt-oss-120b": {
"supports_reasoning": true,
"max_tokens": 131072,
"max_input_tokens": 131072,
"max_output_tokens": 131072,
@ -48665,6 +48666,7 @@
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/openai/gpt-oss-20b": {
"supports_reasoning": true,
"max_tokens": 131072,
"max_input_tokens": 131072,
"max_output_tokens": 131072,
@ -48675,6 +48677,7 @@
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/zai-org/GLM-4.5": {
"supports_reasoning": true,
"max_tokens": 131072,
"max_input_tokens": 131072,
"max_output_tokens": 131072,
@ -48703,6 +48706,7 @@
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/Qwen/Qwen3-235B-A22B-Thinking-2507": {
"supports_reasoning": true,
"max_tokens": 262144,
"max_input_tokens": 262144,
"max_output_tokens": 262144,
@ -48759,6 +48763,7 @@
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/deepseek-ai/DeepSeek-V3.1": {
"supports_reasoning": true,
"max_tokens": 128000,
"max_input_tokens": 161000,
"max_output_tokens": 128000,
@ -48769,6 +48774,7 @@
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/deepseek-ai/DeepSeek-R1-0528": {
"supports_reasoning": true,
"max_tokens": 161000,
"max_input_tokens": 161000,
"max_output_tokens": 161000,
@ -58630,6 +58636,7 @@
"supports_vision": false
},
"wandb/deepseek-ai/DeepSeek-V4-Flash": {
"supports_reasoning": true,
"max_tokens": 1048576,
"max_input_tokens": 1048576,
"input_cost_per_token": 1.4e-07,
@ -58642,6 +58649,7 @@
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/deepseek-ai/DeepSeek-V4-Flash-0731": {
"supports_reasoning": true,
"max_tokens": 262144,
"max_input_tokens": 262144,
"input_cost_per_token": 1.3e-07,
@ -58654,6 +58662,7 @@
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/deepseek-ai/DeepSeek-V4-Pro": {
"supports_reasoning": true,
"max_tokens": 1048576,
"max_input_tokens": 1048576,
"input_cost_per_token": 1.15e-06,
@ -58666,6 +58675,7 @@
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/google/gemma-4-31B-it": {
"supports_reasoning": true,
"max_tokens": 262144,
"max_input_tokens": 262144,
"input_cost_per_token": 1e-07,
@ -58706,6 +58716,7 @@
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/MiniMaxAI/MiniMax-M3": {
"supports_reasoning": true,
"max_tokens": 262144,
"max_input_tokens": 262144,
"input_cost_per_token": 2.3e-07,
@ -58718,6 +58729,7 @@
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/moonshotai/Kimi-K2.7-Code": {
"supports_reasoning": true,
"max_tokens": 262144,
"max_input_tokens": 262144,
"input_cost_per_token": 7.1e-07,
@ -58730,6 +58742,7 @@
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/moonshotai/Kimi-K2.6": {
"supports_reasoning": true,
"max_tokens": 262144,
"max_input_tokens": 262144,
"input_cost_per_token": 6.5e-07,
@ -58742,6 +58755,7 @@
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B": {
"supports_reasoning": true,
"max_tokens": 262144,
"max_input_tokens": 262144,
"input_cost_per_token": 1e-07,
@ -58754,6 +58768,7 @@
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B": {
"supports_reasoning": true,
"max_tokens": 262144,
"max_input_tokens": 262144,
"input_cost_per_token": 7.5e-07,
@ -58776,6 +58791,7 @@
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/Qwen/Qwen3.8-27B": {
"supports_reasoning": true,
"max_tokens": 262144,
"max_input_tokens": 262144,
"input_cost_per_token": 4e-07,
@ -58788,6 +58804,7 @@
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/Qwen/Qwen3.6-35B-A3B": {
"supports_reasoning": true,
"max_tokens": 262144,
"max_input_tokens": 262144,
"input_cost_per_token": 2.5e-07,
@ -58798,6 +58815,7 @@
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/Qwen/Qwen3.6-27B": {
"supports_reasoning": true,
"max_tokens": 262144,
"max_input_tokens": 262144,
"input_cost_per_token": 6e-07,
@ -58810,6 +58828,7 @@
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/Qwen/Qwen3.5-35B-A3B": {
"supports_reasoning": true,
"max_tokens": 262144,
"max_input_tokens": 262144,
"input_cost_per_token": 2.5e-07,
@ -58829,7 +58848,28 @@
"supports_vision": false,
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/deepseek-ai/DeepSeek-V4-Pro-0813": {
"litellm_provider": "wandb",
"mode": "chat",
"supports_reasoning": true,
"input_cost_per_token": 0.00000131,
"output_cost_per_token": 0.00000396,
"cache_read_input_token_cost": 0.000000044,
"supports_prompt_caching": true,
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/ibm-granite/granite-4.2-8b": {
"litellm_provider": "wandb",
"mode": "chat",
"supports_reasoning": true,
"input_cost_per_token": 0.0000001,
"output_cost_per_token": 0.00000015,
"cache_read_input_token_cost": 0.00000005,
"supports_prompt_caching": true,
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/zai-org/GLM-5.2": {
"supports_reasoning": true,
"max_tokens": 262144,
"max_input_tokens": 262144,
"input_cost_per_token": 7.6e-07,

View file

@ -48655,6 +48655,7 @@
"output_cost_per_token": 0.0
},
"wandb/openai/gpt-oss-120b": {
"supports_reasoning": true,
"max_tokens": 131072,
"max_input_tokens": 131072,
"max_output_tokens": 131072,
@ -48665,6 +48666,7 @@
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/openai/gpt-oss-20b": {
"supports_reasoning": true,
"max_tokens": 131072,
"max_input_tokens": 131072,
"max_output_tokens": 131072,
@ -48675,6 +48677,7 @@
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/zai-org/GLM-4.5": {
"supports_reasoning": true,
"max_tokens": 131072,
"max_input_tokens": 131072,
"max_output_tokens": 131072,
@ -48703,6 +48706,7 @@
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/Qwen/Qwen3-235B-A22B-Thinking-2507": {
"supports_reasoning": true,
"max_tokens": 262144,
"max_input_tokens": 262144,
"max_output_tokens": 262144,
@ -48759,6 +48763,7 @@
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/deepseek-ai/DeepSeek-V3.1": {
"supports_reasoning": true,
"max_tokens": 128000,
"max_input_tokens": 161000,
"max_output_tokens": 128000,
@ -48769,6 +48774,7 @@
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/deepseek-ai/DeepSeek-R1-0528": {
"supports_reasoning": true,
"max_tokens": 161000,
"max_input_tokens": 161000,
"max_output_tokens": 161000,
@ -58630,6 +58636,7 @@
"supports_vision": false
},
"wandb/deepseek-ai/DeepSeek-V4-Flash": {
"supports_reasoning": true,
"max_tokens": 1048576,
"max_input_tokens": 1048576,
"input_cost_per_token": 1.4e-07,
@ -58642,6 +58649,7 @@
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/deepseek-ai/DeepSeek-V4-Flash-0731": {
"supports_reasoning": true,
"max_tokens": 262144,
"max_input_tokens": 262144,
"input_cost_per_token": 1.3e-07,
@ -58654,6 +58662,7 @@
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/deepseek-ai/DeepSeek-V4-Pro": {
"supports_reasoning": true,
"max_tokens": 1048576,
"max_input_tokens": 1048576,
"input_cost_per_token": 1.15e-06,
@ -58666,6 +58675,7 @@
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/google/gemma-4-31B-it": {
"supports_reasoning": true,
"max_tokens": 262144,
"max_input_tokens": 262144,
"input_cost_per_token": 1e-07,
@ -58706,6 +58716,7 @@
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/MiniMaxAI/MiniMax-M3": {
"supports_reasoning": true,
"max_tokens": 262144,
"max_input_tokens": 262144,
"input_cost_per_token": 2.3e-07,
@ -58718,6 +58729,7 @@
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/moonshotai/Kimi-K2.7-Code": {
"supports_reasoning": true,
"max_tokens": 262144,
"max_input_tokens": 262144,
"input_cost_per_token": 7.1e-07,
@ -58730,6 +58742,7 @@
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/moonshotai/Kimi-K2.6": {
"supports_reasoning": true,
"max_tokens": 262144,
"max_input_tokens": 262144,
"input_cost_per_token": 6.5e-07,
@ -58742,6 +58755,7 @@
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B": {
"supports_reasoning": true,
"max_tokens": 262144,
"max_input_tokens": 262144,
"input_cost_per_token": 1e-07,
@ -58754,6 +58768,7 @@
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B": {
"supports_reasoning": true,
"max_tokens": 262144,
"max_input_tokens": 262144,
"input_cost_per_token": 7.5e-07,
@ -58776,6 +58791,7 @@
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/Qwen/Qwen3.8-27B": {
"supports_reasoning": true,
"max_tokens": 262144,
"max_input_tokens": 262144,
"input_cost_per_token": 4e-07,
@ -58788,6 +58804,7 @@
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/Qwen/Qwen3.6-35B-A3B": {
"supports_reasoning": true,
"max_tokens": 262144,
"max_input_tokens": 262144,
"input_cost_per_token": 2.5e-07,
@ -58798,6 +58815,7 @@
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/Qwen/Qwen3.6-27B": {
"supports_reasoning": true,
"max_tokens": 262144,
"max_input_tokens": 262144,
"input_cost_per_token": 6e-07,
@ -58810,6 +58828,7 @@
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/Qwen/Qwen3.5-35B-A3B": {
"supports_reasoning": true,
"max_tokens": 262144,
"max_input_tokens": 262144,
"input_cost_per_token": 2.5e-07,
@ -58829,7 +58848,28 @@
"supports_vision": false,
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/deepseek-ai/DeepSeek-V4-Pro-0813": {
"litellm_provider": "wandb",
"mode": "chat",
"supports_reasoning": true,
"input_cost_per_token": 0.00000131,
"output_cost_per_token": 0.00000396,
"cache_read_input_token_cost": 0.000000044,
"supports_prompt_caching": true,
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/ibm-granite/granite-4.2-8b": {
"litellm_provider": "wandb",
"mode": "chat",
"supports_reasoning": true,
"input_cost_per_token": 0.0000001,
"output_cost_per_token": 0.00000015,
"cache_read_input_token_cost": 0.00000005,
"supports_prompt_caching": true,
"source": "https://wandb.ai/site/pricing/tokens/"
},
"wandb/zai-org/GLM-5.2": {
"supports_reasoning": true,
"max_tokens": 262144,
"max_input_tokens": 262144,
"input_cost_per_token": 7.6e-07,

View file

@ -5,18 +5,92 @@ These tests validate the WandbInferenceConfig class which extends OpenAIGPTConfi
Nebius AI Studio is an OpenAI-compatible provider with minor customizations.
"""
import json
from typing import Final
import pytest
import respx
import litellm
from litellm import completion
from litellm.llms.wandb.chat.transformation import WandbConfig
WANDB_REASONING_MODELS: Final = (
"deepseek-ai/DeepSeek-V4-Flash",
"deepseek-ai/DeepSeek-V4-Flash-0731",
"deepseek-ai/DeepSeek-V4-Pro",
"deepseek-ai/DeepSeek-V4-Pro-0813",
"deepseek-ai/DeepSeek-V3.1",
"google/gemma-4-31B-it",
"ibm-granite/granite-4.2-8b",
"MiniMaxAI/MiniMax-M3",
"moonshotai/Kimi-K2.7-Code",
"moonshotai/Kimi-K2.6",
"nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B",
"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B",
"openai/gpt-oss-120b",
"openai/gpt-oss-20b",
"Qwen/Qwen3.8-27B",
"Qwen/Qwen3.6-35B-A3B",
"Qwen/Qwen3.6-27B",
"Qwen/Qwen3.5-35B-A3B",
"zai-org/GLM-5.2",
"moonshotai/Kimi-K2.5",
"MiniMaxAI/MiniMax-M2.5",
"zai-org/GLM-4.5",
"Qwen/Qwen3-235B-A22B-Thinking-2507",
"deepseek-ai/DeepSeek-R1-0528",
)
@pytest.fixture
def wandb_test_config(local_model_cost_map, monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.setattr(litellm, "disable_aiohttp_transport", True)
monkeypatch.setattr(litellm, "telemetry", False)
monkeypatch.setattr(litellm, "drop_params", False)
@pytest.fixture
def wandb_request_mock(respx_mock: respx.MockRouter) -> respx.Route:
return respx_mock.post("https://api.inference.wandb.ai/v1/chat/completions").respond(
json={
"id": "chatcmpl-123",
"object": "chat.completion",
"created": 1677652288,
"model": "test-model",
"choices": [
{
"index": 0,
"message": {"role": "assistant", "content": "Done"},
"finish_reason": "stop",
}
],
"usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2},
},
status_code=200,
)
class TestWandbConfig:
"""Test class for WandB Inference functionality"""
@pytest.mark.parametrize("model", WANDB_REASONING_MODELS)
def test_map_openai_params_preserves_reasoning_effort(self, wandb_test_config, model: str):
assert litellm.model_cost[f"wandb/{model}"].get("supports_reasoning") is True
supported_params = litellm.get_supported_openai_params(model=f"wandb/{model}")
assert supported_params is not None
assert "reasoning_effort" in supported_params
result = WandbConfig().map_openai_params(
non_default_params={"reasoning_effort": "medium", "max_completion_tokens": 64},
optional_params={},
model=model,
drop_params=True,
)
assert result == {"reasoning_effort": "medium", "max_tokens": 64}
def test_default_api_base(self):
"""Test that default API base is used when none is provided"""
config = WandbConfig()
@ -139,3 +213,80 @@ class TestWandbConfig:
# Check for specific content in the response
assert "```python" in content
assert "Hey from LiteLLM" in content
@pytest.mark.respx()
@pytest.mark.parametrize(
"model,effort",
tuple((model, "medium") for model in WANDB_REASONING_MODELS)
+ (
("Qwen/Qwen3.8-27B", "low"),
("Qwen/Qwen3.8-27B", "xhigh"),
),
)
def test_wandb_completion_preserves_reasoning_effort_with_drop_params(
self, wandb_test_config, wandb_request_mock: respx.Route, model: str, effort: str
):
completion(
model=f"wandb/{model}",
messages=[{"role": "user", "content": "Hello"}],
api_key="fake-wandb-key",
api_base="https://api.inference.wandb.ai/v1",
reasoning_effort=effort,
max_completion_tokens=64,
drop_params=True,
)
assert wandb_request_mock.call_count == 1
request_body = json.loads(wandb_request_mock.calls[0].request.content)
assert request_body["model"] == model
assert request_body["reasoning_effort"] == effort
assert request_body["max_tokens"] == 64
assert "max_completion_tokens" not in request_body
@pytest.mark.respx(assert_all_called=False)
@pytest.mark.parametrize("drop_params", [True, False])
@pytest.mark.parametrize(
"model,explicit_false",
[
("meta-llama/Llama-3.1-8B-Instruct", False),
("unknown-model", False),
("openai/gpt-oss-20b", True),
],
)
def test_wandb_completion_without_reasoning_support(
self,
wandb_test_config,
wandb_request_mock: respx.Route,
respx_mock: respx.MockRouter,
monkeypatch: pytest.MonkeyPatch,
model: str,
explicit_false: bool,
drop_params: bool,
):
with monkeypatch.context() as context:
if explicit_false:
context.setitem(litellm.model_cost[f"wandb/{model}"], "supports_reasoning", False)
kwargs = {
"model": f"wandb/{model}",
"messages": [{"role": "user", "content": "Hello"}],
"api_key": "fake-wandb-key",
"api_base": "https://api.inference.wandb.ai/v1",
"reasoning_effort": "medium",
"drop_params": drop_params,
}
if not drop_params:
with pytest.raises(litellm.UnsupportedParamsError, match="reasoning_effort"):
completion(**kwargs)
assert len(respx_mock.calls) == 0
return
completion(**kwargs)
assert wandb_request_mock.call_count == 1
request_body = json.loads(wandb_request_mock.calls[0].request.content)
assert request_body["model"] == model
assert "reasoning_effort" not in request_body
supported_params = litellm.get_supported_openai_params(model=f"wandb/{model}")
assert supported_params is not None
assert "reasoning_effort" not in supported_params