From e7dba75bf2b7db821a0725c022cbf37b57ab1a68 Mon Sep 17 00:00:00 2001 From: hx <1367557521@qq.com> Date: Tue, 8 Sep 2026 11:16:18 +0000 Subject: [PATCH 1/7] feat: auto-translate thinking and reasoning_effort from model_info When a deployment advertises thinking_send_via, rewrite top-level thinking and reasoning_effort into the vendor shape (usually extra_body) before drop_params can strip them on openai-compatible routes Co-authored-by: HX --- .../thinking_param_translation.py | 199 ++++++++++++++++++ litellm/main.py | 1 + litellm/utils.py | 53 +++++ .../test_thinking_param_translation.py | 141 +++++++++++++ 4 files changed, 394 insertions(+) create mode 100644 litellm/litellm_core_utils/thinking_param_translation.py create mode 100644 tests/unit/litellm_core_utils/test_thinking_param_translation.py diff --git a/litellm/litellm_core_utils/thinking_param_translation.py b/litellm/litellm_core_utils/thinking_param_translation.py new file mode 100644 index 00000000000..c3a183def1b --- /dev/null +++ b/litellm/litellm_core_utils/thinking_param_translation.py @@ -0,0 +1,199 @@ +from __future__ import annotations + +from collections.abc import Mapping, Sequence +from dataclasses import dataclass +from types import MappingProxyType +from typing import Final + +_SEND_VIA_EXTRA_BODY: Final = "extra_body" +_SEND_VIA_PROVIDER_MAPPED: Final = "provider_mapped" + +_EFFORT_FALLBACKS: Final[Mapping[str, tuple[str, ...]]] = MappingProxyType( + { + "xhigh": ("max", "high"), + "max": ("xhigh", "high"), + "medium": ("high", "low"), + "minimal": ("low", "none"), + "none": ("low",), + } +) + + +@dataclass(frozen=True, slots=True) +class ThinkingParamsState: + thinking: object | None + reasoning_effort: object | None + extra_body: Mapping[str, object] + + +def _as_str_tuple(value: object) -> tuple[str, ...]: + if not isinstance(value, (list, tuple)): + return () + return tuple(item for item in value if isinstance(item, str)) + + +def _thinking_enabled(thinking: object) -> bool: + if isinstance(thinking, bool): + return thinking + if isinstance(thinking, str): + return thinking.lower() in {"enabled", "true", "1", "auto"} + if isinstance(thinking, Mapping): + typ: Final = thinking.get("type") + if isinstance(typ, str): + return typ.lower() in {"enabled", "auto", "true"} + enabled: Final = thinking.get("enabled") + if isinstance(enabled, bool): + return enabled + return False + + +def _thinking_type_value(thinking: object, allowed: Sequence[str]) -> str | None: + if isinstance(thinking, str): + candidate: Final = thinking + elif isinstance(thinking, Mapping): + raw: Final = thinking.get("type") + candidate = raw if isinstance(raw, str) else None + elif isinstance(thinking, bool): + candidate = "enabled" if thinking else "disabled" + else: + candidate = None + if candidate is None: + return None + if not allowed or candidate in allowed: + return candidate + if candidate == "auto" and "enabled" in allowed: + return "enabled" + return None + + +def _thinking_payload(thinking: object, typ: str) -> Mapping[str, object]: + if not isinstance(thinking, Mapping): + return MappingProxyType({"type": typ}) + budget: Final = thinking.get("budget_tokens") + if isinstance(budget, int): + return MappingProxyType({"type": typ, "budget_tokens": budget}) + return MappingProxyType({"type": typ}) + + +def _clamp_effort(value: object, allowed: Sequence[str]) -> str | None: + if not isinstance(value, str): + return None + if not allowed: + return None + if value in allowed: + return value + for fallback in _EFFORT_FALLBACKS.get(value, ()): + if fallback in allowed: + return fallback + return None + + +def _map_thinking_to_extra_body( + *, + thinking_param: str | None, + thinking: object, + thinking_values: Sequence[str], +) -> Mapping[str, object]: + match thinking_param: + case "thinking.type": + typ: Final = _thinking_type_value(thinking, thinking_values) + if typ is None: + return MappingProxyType({}) + return MappingProxyType({"thinking": dict(_thinking_payload(thinking, typ))}) + case "thinking": + if isinstance(thinking, Mapping): + return MappingProxyType({"thinking": dict(thinking)}) + typ_only: Final = _thinking_type_value(thinking, thinking_values or ("enabled", "disabled")) + if typ_only is None: + return MappingProxyType({}) + return MappingProxyType({"thinking": {"type": typ_only}}) + case "enable_thinking": + return MappingProxyType({"enable_thinking": _thinking_enabled(thinking)}) + case "chat_template_kwargs": + return MappingProxyType({"chat_template_kwargs": {"enable_thinking": _thinking_enabled(thinking)}}) + case None: + return MappingProxyType({}) + case _: + return MappingProxyType({}) + + +def translate_thinking_params( + *, + model_info: Mapping[str, object] | None, + state: ThinkingParamsState, +) -> ThinkingParamsState: + if model_info is None: + return state + + send_via: Final = model_info.get("thinking_send_via") + if send_via not in {_SEND_VIA_EXTRA_BODY, _SEND_VIA_PROVIDER_MAPPED}: + return state + + supports_reasoning: Final = model_info.get("supports_reasoning") is True + thinking_param_raw: Final = model_info.get("thinking_param") + thinking_param: Final = thinking_param_raw if isinstance(thinking_param_raw, str) else None + thinking_values: Final = _as_str_tuple(model_info.get("thinking_values")) + effort_values: Final = _as_str_tuple(model_info.get("reasoning_effort_values")) + + if not supports_reasoning and send_via != _SEND_VIA_PROVIDER_MAPPED: + return state + + thinking: Final = state.thinking + effort: Final = state.reasoning_effort + if thinking is None and effort is None: + return state + + existing_extra: Final = dict(state.extra_body) + patch: dict[str, object] = {} + keep_thinking: Final = send_via == _SEND_VIA_PROVIDER_MAPPED + + if thinking is not None and send_via == _SEND_VIA_EXTRA_BODY: + patch.update( + _map_thinking_to_extra_body( + thinking_param=thinking_param, + thinking=thinking, + thinking_values=thinking_values, + ) + ) + + if effort is not None: + clamped: Final = _clamp_effort(effort, effort_values) + if clamped is not None: + patch["reasoning_effort"] = clamped + elif not effort_values and send_via == _SEND_VIA_EXTRA_BODY and isinstance(effort, str): + patch["reasoning_effort"] = effort + + if not patch: + return state + + thinking_mapped: Final = any(key in patch for key in ("thinking", "enable_thinking", "chat_template_kwargs")) + next_thinking: Final = thinking if (keep_thinking or not thinking_mapped) else None + next_effort: Final = None if "reasoning_effort" in patch else effort + merged_extra: Final = MappingProxyType({**existing_extra, **patch}) + return ThinkingParamsState( + thinking=next_thinking, + reasoning_effort=next_effort, + extra_body=merged_extra, + ) + + +def apply_thinking_param_translation( + *, + model_info: Mapping[str, object] | None, + thinking: object | None, + reasoning_effort: object | None, + existing_extra_body: Mapping[str, object] | None, +) -> ThinkingParamsState: + base_extra: Final = ( + MappingProxyType(dict(existing_extra_body)) + if isinstance(existing_extra_body, Mapping) + else MappingProxyType({}) + ) + return translate_thinking_params( + model_info=model_info, + state=ThinkingParamsState( + thinking=thinking, + reasoning_effort=reasoning_effort, + extra_body=base_extra, + ), + ) diff --git a/litellm/main.py b/litellm/main.py index 6c85adf3ae8..a125c792846 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -5621,6 +5621,7 @@ def completion( "prompt_cache_key": prompt_cache_key, "allowed_openai_params": allowed_openai_params, "base_model": base_model, + "model_info": model_info if isinstance(model_info, dict) else None, } optional_params = get_optional_params(**optional_param_args, **non_default_params) processed_non_default_params: Final = pre_process_non_default_params( diff --git a/litellm/utils.py b/litellm/utils.py index d45b29c0f16..866fe4a9cd9 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -294,6 +294,9 @@ from typing import TYPE_CHECKING, Any, Final, Literal, Protocol, cast, runtime_c from typing_extensions import assert_never from litellm import utils as litellm_utils +from litellm.litellm_core_utils.thinking_param_translation import ( + apply_thinking_param_translation, +) # These are lazy loaded via __getattr__ from litellm.llms.base_llm.base_utils import ( @@ -4321,6 +4324,49 @@ def remove_sensitive_keys_from_dict(d: dict) -> dict: return d +def _apply_model_info_thinking_translation( + *, + model_info: Mapping[str, object] | None, + passed_params: dict, + non_default_params: dict, +) -> None: + prior_thinking: Final = non_default_params.get("thinking", passed_params.get("thinking")) + prior_effort: Final = non_default_params.get("reasoning_effort", passed_params.get("reasoning_effort")) + existing_extra_raw: Final = passed_params.get("extra_body") + existing_extra: Final = existing_extra_raw if isinstance(existing_extra_raw, Mapping) else None + translated: Final = apply_thinking_param_translation( + model_info=model_info, + thinking=prior_thinking, + reasoning_effort=prior_effort, + existing_extra_body=existing_extra, + ) + prior_extra: Final = dict(existing_extra) if existing_extra is not None else {} + if ( + translated.thinking is prior_thinking + and translated.reasoning_effort is prior_effort + and dict(translated.extra_body) == prior_extra + ): + return + + # mutable-ok: get_optional_params already mutates passed_params / non_default_params in place + if translated.thinking is None: + non_default_params.pop("thinking", None) + passed_params["thinking"] = None + else: + non_default_params["thinking"] = translated.thinking + passed_params["thinking"] = translated.thinking + + if translated.reasoning_effort is None: + non_default_params.pop("reasoning_effort", None) + passed_params["reasoning_effort"] = None + else: + non_default_params["reasoning_effort"] = translated.reasoning_effort + passed_params["reasoning_effort"] = translated.reasoning_effort + + if translated.extra_body: + passed_params["extra_body"] = dict(translated.extra_body) + + def pre_process_optional_params(passed_params: dict, non_default_params: dict, custom_llm_provider: str) -> dict: """For .completion(), preprocess optional params""" optional_params: dict = {} @@ -4443,6 +4489,7 @@ def get_optional_params( store: bool | None = None, prompt_cache_key: str | None = None, base_model: str | None = None, + model_info: Mapping[str, object] | None = None, **kwargs, ): drop_params = normalize_drop_params(drop_params) # rebind-ok: config and DB deployments pass "true" as a string @@ -4452,6 +4499,7 @@ def get_optional_params( # non_default_params / _check_valid_arg — it's a routing hint, not an # OpenAI param. passed_params.pop("base_model", None) + model_info_for_translation: Final = passed_params.pop("model_info", None) provider_config: BaseConfig | None = None if custom_llm_provider is not None and custom_llm_provider in [provider.value for provider in LlmProviders]: provider_config = ProviderConfigManager.get_provider_chat_config( @@ -4467,6 +4515,11 @@ def get_optional_params( model=model, provider_config=provider_config, ) + _apply_model_info_thinking_translation( + model_info=model_info_for_translation if isinstance(model_info_for_translation, Mapping) else None, + passed_params=passed_params, + non_default_params=non_default_params, + ) optional_params = pre_process_optional_params( passed_params=passed_params, non_default_params=non_default_params, diff --git a/tests/unit/litellm_core_utils/test_thinking_param_translation.py b/tests/unit/litellm_core_utils/test_thinking_param_translation.py new file mode 100644 index 00000000000..49ebbff9379 --- /dev/null +++ b/tests/unit/litellm_core_utils/test_thinking_param_translation.py @@ -0,0 +1,141 @@ +from types import MappingProxyType + +from litellm.litellm_core_utils.thinking_param_translation import ( + ThinkingParamsState, + apply_thinking_param_translation, + translate_thinking_params, +) +from litellm.utils import get_optional_params + + +def _extra_body_model_info(**overrides: object) -> dict[str, object]: + base: dict[str, object] = { + "supports_reasoning": True, + "thinking_param": "thinking.type", + "thinking_values": ["enabled", "disabled"], + "reasoning_effort_values": ["low", "high", "max"], + "thinking_send_via": "extra_body", + } + return {**base, **overrides} + + +def test_translate_thinking_type_and_effort_to_extra_body(): + result = apply_thinking_param_translation( + model_info=_extra_body_model_info(), + thinking={"type": "enabled", "budget_tokens": 1024}, + reasoning_effort="high", + existing_extra_body=None, + ) + assert result.thinking is None + assert result.reasoning_effort is None + assert dict(result.extra_body) == { + "thinking": {"type": "enabled", "budget_tokens": 1024}, + "reasoning_effort": "high", + } + + +def test_translate_enable_thinking_bool(): + result = apply_thinking_param_translation( + model_info=_extra_body_model_info( + thinking_param="enable_thinking", + thinking_values=["true", "false"], + ), + thinking={"type": "enabled"}, + reasoning_effort=None, + existing_extra_body=None, + ) + assert result.thinking is None + assert dict(result.extra_body) == {"enable_thinking": True} + + +def test_translate_chat_template_kwargs(): + result = apply_thinking_param_translation( + model_info=_extra_body_model_info( + thinking_param="chat_template_kwargs", + thinking_values=[], + reasoning_effort_values=["low", "medium", "high"], + ), + thinking={"type": "disabled"}, + reasoning_effort="medium", + existing_extra_body=None, + ) + assert dict(result.extra_body) == { + "chat_template_kwargs": {"enable_thinking": False}, + "reasoning_effort": "medium", + } + + +def test_translate_clamps_effort_aliases(): + result = apply_thinking_param_translation( + model_info=_extra_body_model_info(reasoning_effort_values=["low", "high", "max"]), + thinking=None, + reasoning_effort="xhigh", + existing_extra_body=None, + ) + assert result.reasoning_effort is None + assert result.extra_body["reasoning_effort"] == "max" + + +def test_translate_provider_mapped_keeps_thinking_moves_effort(): + result = translate_thinking_params( + model_info=_extra_body_model_info(thinking_send_via="provider_mapped"), + state=ThinkingParamsState( + thinking={"type": "enabled"}, + reasoning_effort="high", + extra_body=MappingProxyType({}), + ), + ) + assert result.thinking == {"type": "enabled"} + assert result.reasoning_effort is None + assert dict(result.extra_body) == {"reasoning_effort": "high"} + + +def test_translate_noop_without_model_info(): + state = ThinkingParamsState( + thinking={"type": "enabled"}, + reasoning_effort="high", + extra_body=MappingProxyType({}), + ) + assert translate_thinking_params(model_info=None, state=state) is state + + +def test_translate_noop_when_send_via_na(): + result = apply_thinking_param_translation( + model_info=_extra_body_model_info(thinking_send_via="n/a"), + thinking={"type": "enabled"}, + reasoning_effort="high", + existing_extra_body=None, + ) + assert result.thinking == {"type": "enabled"} + assert result.reasoning_effort == "high" + assert dict(result.extra_body) == {} + + +def test_get_optional_params_openai_drop_translates_via_model_info(): + optional_params = get_optional_params( + model="deepseek-v4-flash", + custom_llm_provider="openai", + drop_params=True, + thinking={"type": "enabled"}, + reasoning_effort="high", + model_info=_extra_body_model_info(), + ) + assert optional_params.get("thinking") is None + assert optional_params.get("reasoning_effort") is None + assert optional_params["extra_body"]["thinking"] == {"type": "enabled"} + assert optional_params["extra_body"]["reasoning_effort"] == "high" + + +def test_get_optional_params_openai_drop_without_model_info_drops_params(): + optional_params = get_optional_params( + model="gpt-4o", + custom_llm_provider="openai", + drop_params=True, + thinking={"type": "enabled"}, + reasoning_effort="high", + ) + extra_body = optional_params.get("extra_body") or {} + assert "thinking" not in extra_body + assert "reasoning_effort" not in extra_body + assert optional_params.get("thinking") is None + assert optional_params.get("reasoning_effort") is None From 5c2acd480b8ba0f533208361554b952155312a80 Mon Sep 17 00:00:00 2001 From: hx <1367557521@qq.com> Date: Wed, 9 Sep 2026 15:13:22 +0800 Subject: [PATCH 2/7] fix: preserve nested thinking extras and honor drop_params --- litellm/batch_completion/main.py | 24 +++++- .../thinking_param_translation.py | 75 ++++++++++++++----- litellm/utils.py | 4 +- .../test_thinking_param_translation.py | 60 +++++++++++++++ 4 files changed, 140 insertions(+), 23 deletions(-) diff --git a/litellm/batch_completion/main.py b/litellm/batch_completion/main.py index 702dd194fda..7ec71dcd4bb 100644 --- a/litellm/batch_completion/main.py +++ b/litellm/batch_completion/main.py @@ -1,13 +1,29 @@ +from collections.abc import Mapping from concurrent.futures import FIRST_COMPLETED, ThreadPoolExecutor, wait from typing import Final import litellm from litellm._logging import print_verbose -from litellm.utils import get_optional_params +from litellm.utils import get_model_info, get_optional_params from ..llms.vllm.completion import handler as vllm_handler +def _model_info_for_batch( + *, + model: str, + custom_llm_provider: str | None, + kwargs: Mapping[str, object], +) -> Mapping[str, object] | None: + from_kwargs: Final = kwargs.get("model_info") + if isinstance(from_kwargs, Mapping): + return from_kwargs + try: + return get_model_info(model=model, custom_llm_provider=custom_llm_provider) + except Exception: # noqa: BLE001 # get_model_info raises Exception for unmapped models + return None + + def batch_completion( model: str, # Optional OpenAI params: see https://platform.openai.com/docs/api-reference/chat/create @@ -79,9 +95,13 @@ def batch_completion( frequency_penalty=frequency_penalty, logit_bias=logit_bias, user=user, - # params to identify the model model=model, custom_llm_provider=custom_llm_provider, + model_info=_model_info_for_batch( + model=model, + custom_llm_provider=custom_llm_provider, + kwargs=kwargs, + ), ) results = vllm_handler.batch_completions( model=model, diff --git a/litellm/litellm_core_utils/thinking_param_translation.py b/litellm/litellm_core_utils/thinking_param_translation.py index c3a183def1b..9e4beeff231 100644 --- a/litellm/litellm_core_utils/thinking_param_translation.py +++ b/litellm/litellm_core_utils/thinking_param_translation.py @@ -88,6 +88,29 @@ def _clamp_effort(value: object, allowed: Sequence[str]) -> str | None: return None +def _mapping_or_none(value: object) -> Mapping[str, object] | None: + if isinstance(value, Mapping): + return value + return None + + +def _merged_mapping_value(left: Mapping[str, object], right: Mapping[str, object], key: str) -> object: + if key not in right: + return left[key] + if key not in left: + return right[key] + left_map: Final = _mapping_or_none(left[key]) + right_map: Final = _mapping_or_none(right[key]) + if left_map is not None and right_map is not None: + return _deep_merge_pair(left_map, right_map) + return right[key] + + +def _deep_merge_pair(left: Mapping[str, object], right: Mapping[str, object]) -> Mapping[str, object]: + keys: Final = frozenset(left) | frozenset(right) + return MappingProxyType({key: _merged_mapping_value(left, right, key) for key in keys}) + + def _map_thinking_to_extra_body( *, thinking_param: str | None, @@ -117,6 +140,20 @@ def _map_thinking_to_extra_body( return MappingProxyType({}) +def _map_effort_to_extra_body( + *, + effort: object, + effort_values: Sequence[str], + send_via: object, +) -> Mapping[str, object]: + clamped: Final = _clamp_effort(effort, effort_values) + if clamped is not None: + return MappingProxyType({"reasoning_effort": clamped}) + if not effort_values and send_via == _SEND_VIA_EXTRA_BODY and isinstance(effort, str): + return MappingProxyType({"reasoning_effort": effort}) + return MappingProxyType({}) + + def translate_thinking_params( *, model_info: Mapping[str, object] | None, @@ -143,33 +180,33 @@ def translate_thinking_params( if thinking is None and effort is None: return state - existing_extra: Final = dict(state.extra_body) - patch: dict[str, object] = {} keep_thinking: Final = send_via == _SEND_VIA_PROVIDER_MAPPED - - if thinking is not None and send_via == _SEND_VIA_EXTRA_BODY: - patch.update( - _map_thinking_to_extra_body( - thinking_param=thinking_param, - thinking=thinking, - thinking_values=thinking_values, - ) + thinking_patch: Final = ( + _map_thinking_to_extra_body( + thinking_param=thinking_param, + thinking=thinking, + thinking_values=thinking_values, ) - - if effort is not None: - clamped: Final = _clamp_effort(effort, effort_values) - if clamped is not None: - patch["reasoning_effort"] = clamped - elif not effort_values and send_via == _SEND_VIA_EXTRA_BODY and isinstance(effort, str): - patch["reasoning_effort"] = effort - + if thinking is not None and send_via == _SEND_VIA_EXTRA_BODY + else MappingProxyType({}) + ) + effort_patch: Final = ( + _map_effort_to_extra_body( + effort=effort, + effort_values=effort_values, + send_via=send_via, + ) + if effort is not None + else MappingProxyType({}) + ) + patch: Final = MappingProxyType({**thinking_patch, **effort_patch}) if not patch: return state thinking_mapped: Final = any(key in patch for key in ("thinking", "enable_thinking", "chat_template_kwargs")) next_thinking: Final = thinking if (keep_thinking or not thinking_mapped) else None next_effort: Final = None if "reasoning_effort" in patch else effort - merged_extra: Final = MappingProxyType({**existing_extra, **patch}) + merged_extra: Final = _deep_merge_pair(state.extra_body, patch) return ThinkingParamsState( thinking=next_thinking, reasoning_effort=next_effort, diff --git a/litellm/utils.py b/litellm/utils.py index 866fe4a9cd9..bbf40b17468 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -4330,8 +4330,8 @@ def _apply_model_info_thinking_translation( passed_params: dict, non_default_params: dict, ) -> None: - prior_thinking: Final = non_default_params.get("thinking", passed_params.get("thinking")) - prior_effort: Final = non_default_params.get("reasoning_effort", passed_params.get("reasoning_effort")) + prior_thinking: Final = non_default_params.get("thinking") + prior_effort: Final = non_default_params.get("reasoning_effort") existing_extra_raw: Final = passed_params.get("extra_body") existing_extra: Final = existing_extra_raw if isinstance(existing_extra_raw, Mapping) else None translated: Final = apply_thinking_param_translation( diff --git a/tests/unit/litellm_core_utils/test_thinking_param_translation.py b/tests/unit/litellm_core_utils/test_thinking_param_translation.py index 49ebbff9379..c5c91181278 100644 --- a/tests/unit/litellm_core_utils/test_thinking_param_translation.py +++ b/tests/unit/litellm_core_utils/test_thinking_param_translation.py @@ -1,3 +1,4 @@ +import importlib from types import MappingProxyType from litellm.litellm_core_utils.thinking_param_translation import ( @@ -65,6 +66,20 @@ def test_translate_chat_template_kwargs(): } +def test_translate_chat_template_kwargs_preserves_existing_nested_keys(): + result = apply_thinking_param_translation( + model_info=_extra_body_model_info( + thinking_param="chat_template_kwargs", + thinking_values=[], + ), + thinking={"type": "enabled"}, + reasoning_effort=None, + existing_extra_body={"chat_template_kwargs": {"reasoning_budget": 512}}, + ) + assert result.extra_body["chat_template_kwargs"]["reasoning_budget"] == 512 + assert result.extra_body["chat_template_kwargs"]["enable_thinking"] is True + + def test_translate_clamps_effort_aliases(): result = apply_thinking_param_translation( model_info=_extra_body_model_info(reasoning_effort_values=["low", "high", "max"]), @@ -139,3 +154,48 @@ def test_get_optional_params_openai_drop_without_model_info_drops_params(): assert "reasoning_effort" not in extra_body assert optional_params.get("thinking") is None assert optional_params.get("reasoning_effort") is None + + +def test_get_optional_params_does_not_reintroduce_dropped_thinking(): + optional_params = get_optional_params( + model="deepseek-v4-flash", + custom_llm_provider="openai", + drop_params=True, + thinking={"type": "enabled"}, + additional_drop_params=["thinking"], + model_info=_extra_body_model_info( + thinking_param="chat_template_kwargs", + thinking_values=[], + ), + ) + extra_body = optional_params.get("extra_body") or {} + assert optional_params.get("thinking") is None + assert "enable_thinking" not in extra_body + assert "chat_template_kwargs" not in extra_body + + +def test_batch_completion_vllm_passes_model_info(monkeypatch): + batch_completion_mod = importlib.import_module("litellm.batch_completion.main") + + captured: dict[str, object] = {} + looked_up: dict[str, object] = _extra_body_model_info(thinking_param="enable_thinking") + + def fake_get_optional_params(**kwargs: object) -> dict[str, object]: + captured.update(kwargs) + return {} + + def fake_batch_completions(**kwargs: object) -> list[str]: + return ["ok"] + + def fake_get_model_info(**kwargs: object) -> dict[str, object]: + return looked_up + + monkeypatch.setattr(batch_completion_mod, "get_optional_params", fake_get_optional_params) + monkeypatch.setattr(batch_completion_mod.vllm_handler, "batch_completions", fake_batch_completions) + monkeypatch.setattr(batch_completion_mod, "get_model_info", fake_get_model_info) + + batch_completion_mod.batch_completion( + model="vllm/some-model", + messages=[[{"role": "user", "content": "hi"}]], + ) + assert captured.get("model_info") == looked_up From 88e39c8eb3935b910c6b61e72c0aa79d5b1e1ec0 Mon Sep 17 00:00:00 2001 From: hx <1367557521@qq.com> Date: Wed, 16 Sep 2026 10:00:58 +0800 Subject: [PATCH 3/7] refactor: freeze thinking translation collections for LIT002 --- .../thinking_param_translation.py | 21 ++++++++++++------- litellm/utils.py | 6 +++--- 2 files changed, 16 insertions(+), 11 deletions(-) diff --git a/litellm/litellm_core_utils/thinking_param_translation.py b/litellm/litellm_core_utils/thinking_param_translation.py index 9e4beeff231..9cc490d8fd6 100644 --- a/litellm/litellm_core_utils/thinking_param_translation.py +++ b/litellm/litellm_core_utils/thinking_param_translation.py @@ -7,6 +7,9 @@ from typing import Final _SEND_VIA_EXTRA_BODY: Final = "extra_body" _SEND_VIA_PROVIDER_MAPPED: Final = "provider_mapped" +_SEND_VIA_VALUES: Final = frozenset((_SEND_VIA_EXTRA_BODY, _SEND_VIA_PROVIDER_MAPPED)) +_THINKING_ENABLED_STRINGS: Final = frozenset(("enabled", "true", "1", "auto")) +_THINKING_TYPE_ENABLED: Final = frozenset(("enabled", "auto", "true")) _EFFORT_FALLBACKS: Final[Mapping[str, tuple[str, ...]]] = MappingProxyType( { @@ -36,11 +39,11 @@ def _thinking_enabled(thinking: object) -> bool: if isinstance(thinking, bool): return thinking if isinstance(thinking, str): - return thinking.lower() in {"enabled", "true", "1", "auto"} + return thinking.lower() in _THINKING_ENABLED_STRINGS if isinstance(thinking, Mapping): typ: Final = thinking.get("type") if isinstance(typ, str): - return typ.lower() in {"enabled", "auto", "true"} + return typ.lower() in _THINKING_TYPE_ENABLED enabled: Final = thinking.get("enabled") if isinstance(enabled, bool): return enabled @@ -122,18 +125,20 @@ def _map_thinking_to_extra_body( typ: Final = _thinking_type_value(thinking, thinking_values) if typ is None: return MappingProxyType({}) - return MappingProxyType({"thinking": dict(_thinking_payload(thinking, typ))}) + return MappingProxyType({"thinking": _thinking_payload(thinking, typ)}) case "thinking": if isinstance(thinking, Mapping): - return MappingProxyType({"thinking": dict(thinking)}) + return MappingProxyType({"thinking": MappingProxyType({k: thinking[k] for k in thinking})}) typ_only: Final = _thinking_type_value(thinking, thinking_values or ("enabled", "disabled")) if typ_only is None: return MappingProxyType({}) - return MappingProxyType({"thinking": {"type": typ_only}}) + return MappingProxyType({"thinking": MappingProxyType({"type": typ_only})}) case "enable_thinking": return MappingProxyType({"enable_thinking": _thinking_enabled(thinking)}) case "chat_template_kwargs": - return MappingProxyType({"chat_template_kwargs": {"enable_thinking": _thinking_enabled(thinking)}}) + return MappingProxyType( + {"chat_template_kwargs": MappingProxyType({"enable_thinking": _thinking_enabled(thinking)})} + ) case None: return MappingProxyType({}) case _: @@ -163,7 +168,7 @@ def translate_thinking_params( return state send_via: Final = model_info.get("thinking_send_via") - if send_via not in {_SEND_VIA_EXTRA_BODY, _SEND_VIA_PROVIDER_MAPPED}: + if send_via not in _SEND_VIA_VALUES: return state supports_reasoning: Final = model_info.get("supports_reasoning") is True @@ -222,7 +227,7 @@ def apply_thinking_param_translation( existing_extra_body: Mapping[str, object] | None, ) -> ThinkingParamsState: base_extra: Final = ( - MappingProxyType(dict(existing_extra_body)) + MappingProxyType({k: existing_extra_body[k] for k in existing_extra_body}) if isinstance(existing_extra_body, Mapping) else MappingProxyType({}) ) diff --git a/litellm/utils.py b/litellm/utils.py index bbf40b17468..87e007fad97 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -4340,11 +4340,11 @@ def _apply_model_info_thinking_translation( reasoning_effort=prior_effort, existing_extra_body=existing_extra, ) - prior_extra: Final = dict(existing_extra) if existing_extra is not None else {} + prior_extra: Final = dict(existing_extra) if existing_extra is not None else {} # mutable-ok: equality snapshot if ( translated.thinking is prior_thinking and translated.reasoning_effort is prior_effort - and dict(translated.extra_body) == prior_extra + and dict(translated.extra_body) == prior_extra # mutable-ok: MappingProxyType equality snapshot ): return @@ -4364,7 +4364,7 @@ def _apply_model_info_thinking_translation( passed_params["reasoning_effort"] = translated.reasoning_effort if translated.extra_body: - passed_params["extra_body"] = dict(translated.extra_body) + passed_params["extra_body"] = dict(translated.extra_body) # mutable-ok: openai extra_body is a dict def pre_process_optional_params(passed_params: dict, non_default_params: dict, custom_llm_provider: str) -> dict: From 51b4da6f92217284bbce37249eb15a3123c01f9a Mon Sep 17 00:00:00 2001 From: hx <1367557521@qq.com> Date: Fri, 18 Sep 2026 15:32:00 +0800 Subject: [PATCH 4/7] fix: avoid rebinding Final candidate in thinking type mapping --- litellm/litellm_core_utils/thinking_param_translation.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/litellm/litellm_core_utils/thinking_param_translation.py b/litellm/litellm_core_utils/thinking_param_translation.py index 9cc490d8fd6..f9916ed9a86 100644 --- a/litellm/litellm_core_utils/thinking_param_translation.py +++ b/litellm/litellm_core_utils/thinking_param_translation.py @@ -52,9 +52,9 @@ def _thinking_enabled(thinking: object) -> bool: def _thinking_type_value(thinking: object, allowed: Sequence[str]) -> str | None: if isinstance(thinking, str): - candidate: Final = thinking + candidate = thinking elif isinstance(thinking, Mapping): - raw: Final = thinking.get("type") + raw = thinking.get("type") candidate = raw if isinstance(raw, str) else None elif isinstance(thinking, bool): candidate = "enabled" if thinking else "disabled" From 755efe014d9095da0ec1e07d9188ef0a3f4038dc Mon Sep 17 00:00:00 2001 From: hx <1367557521@qq.com> Date: Sun, 27 Sep 2026 18:09:03 +0800 Subject: [PATCH 5/7] fix: preserve caller thinking options and stop mutating optional params Caller extra_body values now win over translated ones at every nesting level, and thinking.type translation keeps the caller's other thinking keys instead of rebuilding the dict from type and budget_tokens only. The translated extra_body is thawed into plain dicts before it reaches the request, since nested MappingProxyType values made completion() fail with 'Object of type mappingproxy is not JSON serializable'. get_optional_params now receives new passed_params / non_default_params dicts from the translation helper instead of having them rewritten in place Co-authored-by: Cursor --- .../thinking_param_translation.py | 65 +-- litellm/utils.py | 91 ++-- .../test_thinking_param_translation.py | 445 ++++++++++++------ 3 files changed, 370 insertions(+), 231 deletions(-) diff --git a/litellm/litellm_core_utils/thinking_param_translation.py b/litellm/litellm_core_utils/thinking_param_translation.py index f9916ed9a86..642881b21cf 100644 --- a/litellm/litellm_core_utils/thinking_param_translation.py +++ b/litellm/litellm_core_utils/thinking_param_translation.py @@ -50,16 +50,21 @@ def _thinking_enabled(thinking: object) -> bool: return False +def _thinking_type_candidate(thinking: object) -> str | None: + match thinking: + case bool(): + return "enabled" if thinking else "disabled" + case str(): + return thinking + case Mapping(): + raw: Final = thinking.get("type") + return raw if isinstance(raw, str) else None + case _: + return None + + def _thinking_type_value(thinking: object, allowed: Sequence[str]) -> str | None: - if isinstance(thinking, str): - candidate = thinking - elif isinstance(thinking, Mapping): - raw = thinking.get("type") - candidate = raw if isinstance(raw, str) else None - elif isinstance(thinking, bool): - candidate = "enabled" if thinking else "disabled" - else: - candidate = None + candidate: Final = _thinking_type_candidate(thinking) if candidate is None: return None if not allowed or candidate in allowed: @@ -72,10 +77,7 @@ def _thinking_type_value(thinking: object, allowed: Sequence[str]) -> str | None def _thinking_payload(thinking: object, typ: str) -> Mapping[str, object]: if not isinstance(thinking, Mapping): return MappingProxyType({"type": typ}) - budget: Final = thinking.get("budget_tokens") - if isinstance(budget, int): - return MappingProxyType({"type": typ, "budget_tokens": budget}) - return MappingProxyType({"type": typ}) + return MappingProxyType({**thinking, "type": typ}) def _clamp_effort(value: object, allowed: Sequence[str]) -> str | None: @@ -110,10 +112,19 @@ def _merged_mapping_value(left: Mapping[str, object], right: Mapping[str, object def _deep_merge_pair(left: Mapping[str, object], right: Mapping[str, object]) -> Mapping[str, object]: - keys: Final = frozenset(left) | frozenset(right) + keys: Final = (*left, *(key for key in right if key not in left)) return MappingProxyType({key: _merged_mapping_value(left, right, key) for key in keys}) +def _thawed(value: object) -> object: + mapping: Final = _mapping_or_none(value) + return value if mapping is None else thaw_mapping(mapping) + + +def thaw_mapping(mapping: Mapping[str, object]) -> dict[str, object]: # mutable-ok: JSON request body + return {key: _thawed(item) for key, item in mapping.items()} # mutable-ok: JSON request body + + def _map_thinking_to_extra_body( *, thinking_param: str | None, @@ -139,8 +150,6 @@ def _map_thinking_to_extra_body( return MappingProxyType( {"chat_template_kwargs": MappingProxyType({"enable_thinking": _thinking_enabled(thinking)})} ) - case None: - return MappingProxyType({}) case _: return MappingProxyType({}) @@ -211,31 +220,9 @@ def translate_thinking_params( thinking_mapped: Final = any(key in patch for key in ("thinking", "enable_thinking", "chat_template_kwargs")) next_thinking: Final = thinking if (keep_thinking or not thinking_mapped) else None next_effort: Final = None if "reasoning_effort" in patch else effort - merged_extra: Final = _deep_merge_pair(state.extra_body, patch) + merged_extra: Final = _deep_merge_pair(patch, state.extra_body) return ThinkingParamsState( thinking=next_thinking, reasoning_effort=next_effort, extra_body=merged_extra, ) - - -def apply_thinking_param_translation( - *, - model_info: Mapping[str, object] | None, - thinking: object | None, - reasoning_effort: object | None, - existing_extra_body: Mapping[str, object] | None, -) -> ThinkingParamsState: - base_extra: Final = ( - MappingProxyType({k: existing_extra_body[k] for k in existing_extra_body}) - if isinstance(existing_extra_body, Mapping) - else MappingProxyType({}) - ) - return translate_thinking_params( - model_info=model_info, - state=ThinkingParamsState( - thinking=thinking, - reasoning_effort=reasoning_effort, - extra_body=base_extra, - ), - ) diff --git a/litellm/utils.py b/litellm/utils.py index 87e007fad97..0d71c65c3c9 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -295,7 +295,9 @@ from typing_extensions import assert_never from litellm import utils as litellm_utils from litellm.litellm_core_utils.thinking_param_translation import ( - apply_thinking_param_translation, + ThinkingParamsState, + thaw_mapping, + translate_thinking_params, ) # These are lazy loaded via __getattr__ @@ -4324,47 +4326,38 @@ def remove_sensitive_keys_from_dict(d: dict) -> dict: return d -def _apply_model_info_thinking_translation( +def _translate_thinking_in_params( *, model_info: Mapping[str, object] | None, - passed_params: dict, - non_default_params: dict, -) -> None: - prior_thinking: Final = non_default_params.get("thinking") - prior_effort: Final = non_default_params.get("reasoning_effort") - existing_extra_raw: Final = passed_params.get("extra_body") - existing_extra: Final = existing_extra_raw if isinstance(existing_extra_raw, Mapping) else None - translated: Final = apply_thinking_param_translation( - model_info=model_info, - thinking=prior_thinking, - reasoning_effort=prior_effort, - existing_extra_body=existing_extra, + passed_params: dict, # mutable-ok: get_optional_params hands over its legacy mutable params + non_default_params: dict, # mutable-ok: get_optional_params hands over its legacy mutable params +) -> tuple[dict, dict]: # mutable-ok: get_optional_params keeps mutating both copies downstream + existing_extra_body: Final = passed_params.get("extra_body") + state: Final = ThinkingParamsState( + thinking=non_default_params.get("thinking"), + reasoning_effort=non_default_params.get("reasoning_effort"), + extra_body=existing_extra_body if isinstance(existing_extra_body, Mapping) else MappingProxyType({}), + ) + translated: Final = translate_thinking_params(model_info=model_info, state=state) + if translated is state: + return passed_params, non_default_params + thinking_values: Final = MappingProxyType( + {"thinking": translated.thinking, "reasoning_effort": translated.reasoning_effort} + ) + untouched_non_default: Final = MappingProxyType( + {key: value for key, value in non_default_params.items() if key not in thinking_values} + ) + surviving_thinking_values: Final = MappingProxyType( + {key: value for key, value in thinking_values.items() if value is not None} + ) + return ( + { # mutable-ok: get_optional_params keeps mutating passed_params downstream + **passed_params, + **thinking_values, + "extra_body": thaw_mapping(translated.extra_body), + }, + {**untouched_non_default, **surviving_thinking_values}, # mutable-ok: _check_valid_arg pops unsupported keys ) - prior_extra: Final = dict(existing_extra) if existing_extra is not None else {} # mutable-ok: equality snapshot - if ( - translated.thinking is prior_thinking - and translated.reasoning_effort is prior_effort - and dict(translated.extra_body) == prior_extra # mutable-ok: MappingProxyType equality snapshot - ): - return - - # mutable-ok: get_optional_params already mutates passed_params / non_default_params in place - if translated.thinking is None: - non_default_params.pop("thinking", None) - passed_params["thinking"] = None - else: - non_default_params["thinking"] = translated.thinking - passed_params["thinking"] = translated.thinking - - if translated.reasoning_effort is None: - non_default_params.pop("reasoning_effort", None) - passed_params["reasoning_effort"] = None - else: - non_default_params["reasoning_effort"] = translated.reasoning_effort - passed_params["reasoning_effort"] = translated.reasoning_effort - - if translated.extra_body: - passed_params["extra_body"] = dict(translated.extra_body) # mutable-ok: openai extra_body is a dict def pre_process_optional_params(passed_params: dict, non_default_params: dict, custom_llm_provider: str) -> dict: @@ -4493,13 +4486,13 @@ def get_optional_params( **kwargs, ): drop_params = normalize_drop_params(drop_params) # rebind-ok: config and DB deployments pass "true" as a string - passed_params: Final = locals().copy() - special_params: Final = passed_params.pop("kwargs") + untranslated_passed_params: Final = locals().copy() + special_params: Final = untranslated_passed_params.pop("kwargs") # Remove base_model from passed_params so it doesn't interfere with # non_default_params / _check_valid_arg — it's a routing hint, not an # OpenAI param. - passed_params.pop("base_model", None) - model_info_for_translation: Final = passed_params.pop("model_info", None) + untranslated_passed_params.pop("base_model", None) + untranslated_passed_params.pop("model_info", None) provider_config: BaseConfig | None = None if custom_llm_provider is not None and custom_llm_provider in [provider.value for provider in LlmProviders]: provider_config = ProviderConfigManager.get_provider_chat_config( @@ -4507,18 +4500,18 @@ def get_optional_params( provider=LlmProviders(custom_llm_provider), base_model=base_model, ) - non_default_params: Final = pre_process_non_default_params( - passed_params=passed_params, + untranslated_non_default_params: Final = pre_process_non_default_params( + passed_params=untranslated_passed_params, special_params=special_params, custom_llm_provider=custom_llm_provider, additional_drop_params=additional_drop_params, model=model, provider_config=provider_config, ) - _apply_model_info_thinking_translation( - model_info=model_info_for_translation if isinstance(model_info_for_translation, Mapping) else None, - passed_params=passed_params, - non_default_params=non_default_params, + passed_params, non_default_params = _translate_thinking_in_params( + model_info=model_info, + passed_params=untranslated_passed_params, + non_default_params=untranslated_non_default_params, ) optional_params = pre_process_optional_params( passed_params=passed_params, diff --git a/tests/unit/litellm_core_utils/test_thinking_param_translation.py b/tests/unit/litellm_core_utils/test_thinking_param_translation.py index c5c91181278..a8cc0016bbd 100644 --- a/tests/unit/litellm_core_utils/test_thinking_param_translation.py +++ b/tests/unit/litellm_core_utils/test_thinking_param_translation.py @@ -1,147 +1,279 @@ -import importlib +import copy +import json +from collections.abc import Mapping from types import MappingProxyType +from typing import Final +import httpx +import openai +import pytest + +import litellm from litellm.litellm_core_utils.thinking_param_translation import ( ThinkingParamsState, - apply_thinking_param_translation, translate_thinking_params, ) from litellm.utils import get_optional_params - -def _extra_body_model_info(**overrides: object) -> dict[str, object]: - base: dict[str, object] = { +_EXTRA_BODY_MODEL_INFO: Final = MappingProxyType( + { "supports_reasoning": True, "thinking_param": "thinking.type", "thinking_values": ["enabled", "disabled"], "reasoning_effort_values": ["low", "high", "max"], "thinking_send_via": "extra_body", } - return {**base, **overrides} +) - -def test_translate_thinking_type_and_effort_to_extra_body(): - result = apply_thinking_param_translation( - model_info=_extra_body_model_info(), - thinking={"type": "enabled", "budget_tokens": 1024}, - reasoning_effort="high", - existing_extra_body=None, - ) - assert result.thinking is None - assert result.reasoning_effort is None - assert dict(result.extra_body) == { - "thinking": {"type": "enabled", "budget_tokens": 1024}, - "reasoning_effort": "high", +_CHAT_COMPLETION_RESPONSE: Final = MappingProxyType( + { + "id": "chatcmpl-thinking", + "object": "chat.completion", + "created": 0, + "model": "deepseek-v4-flash", + "choices": [{"index": 0, "message": {"role": "assistant", "content": "ok"}, "finish_reason": "stop"}], + "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}, } +) -def test_translate_enable_thinking_bool(): - result = apply_thinking_param_translation( - model_info=_extra_body_model_info( - thinking_param="enable_thinking", - thinking_values=["true", "false"], +def _model_info(**overrides: object) -> Mapping[str, object]: + return MappingProxyType({**_EXTRA_BODY_MODEL_INFO, **overrides}) + + +def _state( + *, + thinking: object = None, + reasoning_effort: object = None, + extra_body: Mapping[str, object] = MappingProxyType({}), +) -> ThinkingParamsState: + return ThinkingParamsState(thinking=thinking, reasoning_effort=reasoning_effort, extra_body=extra_body) + + +def _as_plain(value: object) -> object: + if isinstance(value, Mapping): + return {key: _as_plain(item) for key, item in value.items()} + return value + + +def _recording_client(bodies: list[object]) -> openai.OpenAI: + def respond(request: httpx.Request) -> httpx.Response: + bodies.append(json.loads(request.content)) + return httpx.Response(200, json=dict(_CHAT_COMPLETION_RESPONSE)) + + return openai.OpenAI(api_key="test-key", http_client=httpx.Client(transport=httpx.MockTransport(respond))) + + +@pytest.mark.parametrize( + ("model_info", "thinking", "reasoning_effort", "expected_extra_body"), + [ + pytest.param( + _model_info(), + {"type": "enabled", "budget_tokens": 1024, "clear_thinking": False}, + "high", + { + "thinking": {"type": "enabled", "budget_tokens": 1024, "clear_thinking": False}, + "reasoning_effort": "high", + }, + id="thinking_type_keeps_caller_thinking_keys", ), - thinking={"type": "enabled"}, - reasoning_effort=None, - existing_extra_body=None, - ) - assert result.thinking is None - assert dict(result.extra_body) == {"enable_thinking": True} - - -def test_translate_chat_template_kwargs(): - result = apply_thinking_param_translation( - model_info=_extra_body_model_info( - thinking_param="chat_template_kwargs", - thinking_values=[], - reasoning_effort_values=["low", "medium", "high"], + pytest.param( + _model_info(), + {"type": "auto"}, + None, + {"thinking": {"type": "enabled"}}, + id="thinking_type_auto_falls_back_to_enabled", ), - thinking={"type": "disabled"}, - reasoning_effort="medium", - existing_extra_body=None, - ) - assert dict(result.extra_body) == { - "chat_template_kwargs": {"enable_thinking": False}, - "reasoning_effort": "medium", - } - - -def test_translate_chat_template_kwargs_preserves_existing_nested_keys(): - result = apply_thinking_param_translation( - model_info=_extra_body_model_info( - thinking_param="chat_template_kwargs", - thinking_values=[], + pytest.param( + _model_info(), + False, + None, + {"thinking": {"type": "disabled"}}, + id="thinking_type_from_bool", ), - thinking={"type": "enabled"}, - reasoning_effort=None, - existing_extra_body={"chat_template_kwargs": {"reasoning_budget": 512}}, - ) - assert result.extra_body["chat_template_kwargs"]["reasoning_budget"] == 512 - assert result.extra_body["chat_template_kwargs"]["enable_thinking"] is True - - -def test_translate_clamps_effort_aliases(): - result = apply_thinking_param_translation( - model_info=_extra_body_model_info(reasoning_effort_values=["low", "high", "max"]), - thinking=None, - reasoning_effort="xhigh", - existing_extra_body=None, - ) - assert result.reasoning_effort is None - assert result.extra_body["reasoning_effort"] == "max" - - -def test_translate_provider_mapped_keeps_thinking_moves_effort(): + pytest.param( + _model_info(), + "enabled", + None, + {"thinking": {"type": "enabled"}}, + id="thinking_type_from_string", + ), + pytest.param( + _model_info(thinking_param="thinking", thinking_values=[]), + {"type": "enabled", "budget_tokens": 2048}, + None, + {"thinking": {"type": "enabled", "budget_tokens": 2048}}, + id="thinking_dict_passthrough", + ), + pytest.param( + _model_info(thinking_param="thinking", thinking_values=[]), + True, + None, + {"thinking": {"type": "enabled"}}, + id="thinking_from_bool", + ), + pytest.param( + _model_info(thinking_param="enable_thinking"), + {"type": "enabled"}, + None, + {"enable_thinking": True}, + id="enable_thinking_from_type", + ), + pytest.param( + _model_info(thinking_param="enable_thinking"), + {"enabled": True}, + None, + {"enable_thinking": True}, + id="enable_thinking_from_enabled_flag", + ), + pytest.param( + _model_info(thinking_param="enable_thinking"), + "true", + None, + {"enable_thinking": True}, + id="enable_thinking_from_string", + ), + pytest.param( + _model_info(thinking_param="enable_thinking"), + False, + None, + {"enable_thinking": False}, + id="enable_thinking_from_bool", + ), + pytest.param( + _model_info(thinking_param="enable_thinking"), + 1, + None, + {"enable_thinking": False}, + id="enable_thinking_unrecognized_value_is_disabled", + ), + pytest.param( + _model_info(thinking_param="chat_template_kwargs", reasoning_effort_values=["low", "medium", "high"]), + {"type": "disabled"}, + "medium", + {"chat_template_kwargs": {"enable_thinking": False}, "reasoning_effort": "medium"}, + id="chat_template_kwargs", + ), + pytest.param( + _model_info(reasoning_effort_values=["low", "high", "max"]), + None, + "xhigh", + {"reasoning_effort": "max"}, + id="effort_clamped_to_alias", + ), + pytest.param( + _model_info(reasoning_effort_values=["low"]), + None, + "minimal", + {"reasoning_effort": "low"}, + id="effort_clamped_down", + ), + pytest.param( + _model_info(reasoning_effort_values=[]), + None, + "high", + {"reasoning_effort": "high"}, + id="effort_passthrough_without_allowed_values", + ), + pytest.param( + _model_info(reasoning_effort_values=None), + None, + "high", + {"reasoning_effort": "high"}, + id="effort_passthrough_when_allowed_values_missing", + ), + ], +) +def test_translate_moves_params_into_extra_body( + model_info: Mapping[str, object], + thinking: object, + reasoning_effort: object, + expected_extra_body: dict[str, object], +): result = translate_thinking_params( - model_info=_extra_body_model_info(thinking_send_via="provider_mapped"), - state=ThinkingParamsState( + model_info=model_info, state=_state(thinking=thinking, reasoning_effort=reasoning_effort) + ) + + assert (result.thinking, result.reasoning_effort) == (None, None) + assert _as_plain(result.extra_body) == expected_extra_body + + +@pytest.mark.parametrize( + ("model_info", "thinking", "reasoning_effort"), + [ + pytest.param(None, {"type": "enabled"}, "high", id="no_model_info"), + pytest.param(_model_info(thinking_send_via="n/a"), {"type": "enabled"}, "high", id="send_via_not_applicable"), + pytest.param(_model_info(supports_reasoning=False), {"type": "enabled"}, "high", id="reasoning_unsupported"), + pytest.param(_model_info(), None, None, id="nothing_requested"), + pytest.param(_model_info(thinking_param="unknown"), {"type": "enabled"}, None, id="unknown_thinking_param"), + pytest.param(_model_info(thinking_param=None), {"type": "enabled"}, None, id="thinking_param_missing"), + pytest.param(_model_info(), {"type": "adaptive"}, None, id="thinking_type_not_allowed"), + pytest.param(_model_info(), {"budget_tokens": 1024}, None, id="thinking_type_unreadable"), + pytest.param(_model_info(), 1, None, id="thinking_type_unsupported_value"), + pytest.param( + _model_info(thinking_param="thinking", thinking_values=[]), "adaptive", None, id="thinking_value_unmapped" + ), + pytest.param(_model_info(reasoning_effort_values=["low"]), None, "ultra", id="effort_without_fallback"), + pytest.param(_model_info(), None, 5, id="effort_not_a_string"), + ], +) +def test_translate_returns_state_unchanged_when_nothing_applies( + model_info: Mapping[str, object] | None, + thinking: object, + reasoning_effort: object, +): + state = _state(thinking=thinking, reasoning_effort=reasoning_effort) + + assert translate_thinking_params(model_info=model_info, state=state) is state + + +def test_translate_provider_mapped_keeps_thinking_and_moves_effort(): + thinking = {"type": "enabled"} + + result = translate_thinking_params( + model_info=_model_info(thinking_send_via="provider_mapped", supports_reasoning=False), + state=_state(thinking=thinking, reasoning_effort="high"), + ) + + assert result.thinking is thinking + assert result.reasoning_effort is None + assert _as_plain(result.extra_body) == {"reasoning_effort": "high"} + + +def test_translate_keeps_caller_extra_body_values_over_translated_ones(): + result = translate_thinking_params( + model_info=_model_info(thinking_param="chat_template_kwargs", thinking_values=[]), + state=_state( thinking={"type": "enabled"}, reasoning_effort="high", - extra_body=MappingProxyType({}), + extra_body={"chat_template_kwargs": {"enable_thinking": False, "reasoning_budget": 512}, "top_k": 20}, ), ) - assert result.thinking == {"type": "enabled"} - assert result.reasoning_effort is None - assert dict(result.extra_body) == {"reasoning_effort": "high"} + + assert (result.thinking, result.reasoning_effort) == (None, None) + assert _as_plain(result.extra_body) == { + "chat_template_kwargs": {"enable_thinking": False, "reasoning_budget": 512}, + "reasoning_effort": "high", + "top_k": 20, + } -def test_translate_noop_without_model_info(): - state = ThinkingParamsState( - thinking={"type": "enabled"}, - reasoning_effort="high", - extra_body=MappingProxyType({}), - ) - assert translate_thinking_params(model_info=None, state=state) is state - - -def test_translate_noop_when_send_via_na(): - result = apply_thinking_param_translation( - model_info=_extra_body_model_info(thinking_send_via="n/a"), - thinking={"type": "enabled"}, - reasoning_effort="high", - existing_extra_body=None, - ) - assert result.thinking == {"type": "enabled"} - assert result.reasoning_effort == "high" - assert dict(result.extra_body) == {} - - -def test_get_optional_params_openai_drop_translates_via_model_info(): +def test_get_optional_params_moves_thinking_into_extra_body(): optional_params = get_optional_params( model="deepseek-v4-flash", custom_llm_provider="openai", drop_params=True, thinking={"type": "enabled"}, reasoning_effort="high", - model_info=_extra_body_model_info(), + model_info=_model_info(), ) - assert optional_params.get("thinking") is None - assert optional_params.get("reasoning_effort") is None - assert optional_params["extra_body"]["thinking"] == {"type": "enabled"} - assert optional_params["extra_body"]["reasoning_effort"] == "high" + + assert "thinking" not in optional_params + assert "reasoning_effort" not in optional_params + assert optional_params["extra_body"] == {"thinking": {"type": "enabled"}, "reasoning_effort": "high"} -def test_get_optional_params_openai_drop_without_model_info_drops_params(): +def test_get_optional_params_without_model_info_drops_thinking(): optional_params = get_optional_params( model="gpt-4o", custom_llm_provider="openai", @@ -149,11 +281,9 @@ def test_get_optional_params_openai_drop_without_model_info_drops_params(): thinking={"type": "enabled"}, reasoning_effort="high", ) - extra_body = optional_params.get("extra_body") or {} - assert "thinking" not in extra_body - assert "reasoning_effort" not in extra_body - assert optional_params.get("thinking") is None - assert optional_params.get("reasoning_effort") is None + + assert "thinking" not in optional_params + assert "thinking" not in (optional_params.get("extra_body") or {}) def test_get_optional_params_does_not_reintroduce_dropped_thinking(): @@ -163,39 +293,68 @@ def test_get_optional_params_does_not_reintroduce_dropped_thinking(): drop_params=True, thinking={"type": "enabled"}, additional_drop_params=["thinking"], - model_info=_extra_body_model_info( - thinking_param="chat_template_kwargs", - thinking_values=[], - ), + model_info=_model_info(thinking_param="chat_template_kwargs", thinking_values=[]), ) - extra_body = optional_params.get("extra_body") or {} - assert optional_params.get("thinking") is None - assert "enable_thinking" not in extra_body - assert "chat_template_kwargs" not in extra_body + + assert "thinking" not in optional_params + assert optional_params.get("extra_body") in (None, {}) -def test_batch_completion_vllm_passes_model_info(monkeypatch): - batch_completion_mod = importlib.import_module("litellm.batch_completion.main") +def test_get_optional_params_leaves_caller_extra_body_untouched_and_serializable(): + caller_extra_body = {"chat_template_kwargs": {"reasoning_budget": 512}} - captured: dict[str, object] = {} - looked_up: dict[str, object] = _extra_body_model_info(thinking_param="enable_thinking") - - def fake_get_optional_params(**kwargs: object) -> dict[str, object]: - captured.update(kwargs) - return {} - - def fake_batch_completions(**kwargs: object) -> list[str]: - return ["ok"] - - def fake_get_model_info(**kwargs: object) -> dict[str, object]: - return looked_up - - monkeypatch.setattr(batch_completion_mod, "get_optional_params", fake_get_optional_params) - monkeypatch.setattr(batch_completion_mod.vllm_handler, "batch_completions", fake_batch_completions) - monkeypatch.setattr(batch_completion_mod, "get_model_info", fake_get_model_info) - - batch_completion_mod.batch_completion( - model="vllm/some-model", - messages=[[{"role": "user", "content": "hi"}]], + optional_params = get_optional_params( + model="deepseek-v4-flash", + custom_llm_provider="openai", + thinking={"type": "enabled"}, + extra_body=caller_extra_body, + model_info=_model_info(thinking_param="chat_template_kwargs", thinking_values=[]), ) - assert captured.get("model_info") == looked_up + + expected_extra_body = {"chat_template_kwargs": {"reasoning_budget": 512, "enable_thinking": True}} + assert optional_params["extra_body"] == expected_extra_body + assert json.loads(json.dumps(optional_params["extra_body"])) == expected_extra_body + assert copy.deepcopy(optional_params)["extra_body"] == expected_extra_body + assert caller_extra_body == {"chat_template_kwargs": {"reasoning_budget": 512}} + + +def test_completion_sends_translated_thinking_on_the_wire(): + bodies: list[object] = [] + + litellm.completion( + model="openai/deepseek-v4-flash", + messages=[{"role": "user", "content": "hi"}], + thinking={"type": "enabled", "budget_tokens": 1024}, + reasoning_effort="high", + model_info=dict(_model_info()), + client=_recording_client(bodies), + num_retries=0, + ) + + assert bodies == [ + { + "model": "deepseek-v4-flash", + "messages": [{"role": "user", "content": "hi"}], + "thinking": {"type": "enabled", "budget_tokens": 1024}, + "reasoning_effort": "high", + } + ] + + +def test_batch_completion_translates_every_request_like_completion(): + bodies: list[object] = [] + + litellm.batch_completion( + model="openai/deepseek-v4-flash", + messages=[[{"role": "user", "content": "one"}], [{"role": "user", "content": "two"}]], + thinking={"type": "enabled"}, + model_info=dict(_model_info(thinking_param="enable_thinking")), + client=_recording_client(bodies), + num_retries=0, + max_workers=1, + ) + + assert sorted(bodies, key=lambda body: json.dumps(body, sort_keys=True)) == [ + {"model": "deepseek-v4-flash", "messages": [{"role": "user", "content": "one"}], "enable_thinking": True}, + {"model": "deepseek-v4-flash", "messages": [{"role": "user", "content": "two"}], "enable_thinking": True}, + ] From e18ff4b3b73be24313417a39b8cfeede7cca561f Mon Sep 17 00:00:00 2001 From: hx <1367557521@qq.com> Date: Sun, 27 Sep 2026 18:09:03 +0800 Subject: [PATCH 6/7] fix: drop inert model_info lookup from vLLM batch path The vLLM batch path never forwards thinking or reasoning_effort to get_optional_params, so the lookup could not change the request. It also fell back to get_model_info, which completion() does not do. Non-vLLM batch requests already go through completion() with the caller's model_info Co-authored-by: Cursor --- litellm/batch_completion/main.py | 24 ++---------------------- 1 file changed, 2 insertions(+), 22 deletions(-) diff --git a/litellm/batch_completion/main.py b/litellm/batch_completion/main.py index 7ec71dcd4bb..702dd194fda 100644 --- a/litellm/batch_completion/main.py +++ b/litellm/batch_completion/main.py @@ -1,29 +1,13 @@ -from collections.abc import Mapping from concurrent.futures import FIRST_COMPLETED, ThreadPoolExecutor, wait from typing import Final import litellm from litellm._logging import print_verbose -from litellm.utils import get_model_info, get_optional_params +from litellm.utils import get_optional_params from ..llms.vllm.completion import handler as vllm_handler -def _model_info_for_batch( - *, - model: str, - custom_llm_provider: str | None, - kwargs: Mapping[str, object], -) -> Mapping[str, object] | None: - from_kwargs: Final = kwargs.get("model_info") - if isinstance(from_kwargs, Mapping): - return from_kwargs - try: - return get_model_info(model=model, custom_llm_provider=custom_llm_provider) - except Exception: # noqa: BLE001 # get_model_info raises Exception for unmapped models - return None - - def batch_completion( model: str, # Optional OpenAI params: see https://platform.openai.com/docs/api-reference/chat/create @@ -95,13 +79,9 @@ def batch_completion( frequency_penalty=frequency_penalty, logit_bias=logit_bias, user=user, + # params to identify the model model=model, custom_llm_provider=custom_llm_provider, - model_info=_model_info_for_batch( - model=model, - custom_llm_provider=custom_llm_provider, - kwargs=kwargs, - ), ) results = vllm_handler.batch_completions( model=model, From 731bf469945364b034a6170d566709f52ae0268a Mon Sep 17 00:00:00 2001 From: hx <1367557521@qq.com> Date: Sun, 27 Sep 2026 18:28:54 +0800 Subject: [PATCH 7/7] refactor: validate thinking translation inputs instead of narrowing to unknown types isinstance(x, Mapping) narrowed object values to Mapping[Unknown, Unknown], which left the new module and its get_optional_params helper with strict basedpyright errors. Validate through pydantic TypeAdapters instead so both are fully typed Co-authored-by: Cursor --- .../thinking_param_translation.py | 74 +++++++++++-------- litellm/utils.py | 10 +-- .../test_thinking_param_translation.py | 1 + 3 files changed, 48 insertions(+), 37 deletions(-) diff --git a/litellm/litellm_core_utils/thinking_param_translation.py b/litellm/litellm_core_utils/thinking_param_translation.py index 642881b21cf..0ebdb5b3667 100644 --- a/litellm/litellm_core_utils/thinking_param_translation.py +++ b/litellm/litellm_core_utils/thinking_param_translation.py @@ -5,6 +5,12 @@ from dataclasses import dataclass from types import MappingProxyType from typing import Final +from pydantic import TypeAdapter, ValidationError + +_STR_KEYED_MAPPING: Final = TypeAdapter(Mapping[str, object]) +_OBJECT_TUPLE: Final = TypeAdapter(tuple[object, ...]) +_EMPTY: Final[Mapping[str, object]] = MappingProxyType({}) + _SEND_VIA_EXTRA_BODY: Final = "extra_body" _SEND_VIA_PROVIDER_MAPPED: Final = "provider_mapped" _SEND_VIA_VALUES: Final = frozenset((_SEND_VIA_EXTRA_BODY, _SEND_VIA_PROVIDER_MAPPED)) @@ -29,10 +35,19 @@ class ThinkingParamsState: extra_body: Mapping[str, object] +def str_keyed_mapping_or_none(value: object) -> Mapping[str, object] | None: + if not isinstance(value, Mapping): + return None + try: + return _STR_KEYED_MAPPING.validate_python(value) + except ValidationError: + return None + + def _as_str_tuple(value: object) -> tuple[str, ...]: if not isinstance(value, (list, tuple)): return () - return tuple(item for item in value if isinstance(item, str)) + return tuple(item for item in _OBJECT_TUPLE.validate_python(value) if isinstance(item, str)) def _thinking_enabled(thinking: object) -> bool: @@ -40,14 +55,14 @@ def _thinking_enabled(thinking: object) -> bool: return thinking if isinstance(thinking, str): return thinking.lower() in _THINKING_ENABLED_STRINGS - if isinstance(thinking, Mapping): - typ: Final = thinking.get("type") - if isinstance(typ, str): - return typ.lower() in _THINKING_TYPE_ENABLED - enabled: Final = thinking.get("enabled") - if isinstance(enabled, bool): - return enabled - return False + mapping: Final = str_keyed_mapping_or_none(thinking) + if mapping is None: + return False + typ: Final = mapping.get("type") + if isinstance(typ, str): + return typ.lower() in _THINKING_TYPE_ENABLED + enabled: Final = mapping.get("enabled") + return enabled if isinstance(enabled, bool) else False def _thinking_type_candidate(thinking: object) -> str | None: @@ -56,11 +71,10 @@ def _thinking_type_candidate(thinking: object) -> str | None: return "enabled" if thinking else "disabled" case str(): return thinking - case Mapping(): - raw: Final = thinking.get("type") - return raw if isinstance(raw, str) else None case _: - return None + mapping: Final = str_keyed_mapping_or_none(thinking) + raw: Final = mapping.get("type") if mapping is not None else None + return raw if isinstance(raw, str) else None def _thinking_type_value(thinking: object, allowed: Sequence[str]) -> str | None: @@ -75,9 +89,10 @@ def _thinking_type_value(thinking: object, allowed: Sequence[str]) -> str | None def _thinking_payload(thinking: object, typ: str) -> Mapping[str, object]: - if not isinstance(thinking, Mapping): + mapping: Final = str_keyed_mapping_or_none(thinking) + if mapping is None: return MappingProxyType({"type": typ}) - return MappingProxyType({**thinking, "type": typ}) + return MappingProxyType({**mapping, "type": typ}) def _clamp_effort(value: object, allowed: Sequence[str]) -> str | None: @@ -93,19 +108,13 @@ def _clamp_effort(value: object, allowed: Sequence[str]) -> str | None: return None -def _mapping_or_none(value: object) -> Mapping[str, object] | None: - if isinstance(value, Mapping): - return value - return None - - def _merged_mapping_value(left: Mapping[str, object], right: Mapping[str, object], key: str) -> object: if key not in right: return left[key] if key not in left: return right[key] - left_map: Final = _mapping_or_none(left[key]) - right_map: Final = _mapping_or_none(right[key]) + left_map: Final = str_keyed_mapping_or_none(left[key]) + right_map: Final = str_keyed_mapping_or_none(right[key]) if left_map is not None and right_map is not None: return _deep_merge_pair(left_map, right_map) return right[key] @@ -117,7 +126,7 @@ def _deep_merge_pair(left: Mapping[str, object], right: Mapping[str, object]) -> def _thawed(value: object) -> object: - mapping: Final = _mapping_or_none(value) + mapping: Final = str_keyed_mapping_or_none(value) return value if mapping is None else thaw_mapping(mapping) @@ -135,14 +144,15 @@ def _map_thinking_to_extra_body( case "thinking.type": typ: Final = _thinking_type_value(thinking, thinking_values) if typ is None: - return MappingProxyType({}) + return _EMPTY return MappingProxyType({"thinking": _thinking_payload(thinking, typ)}) case "thinking": - if isinstance(thinking, Mapping): - return MappingProxyType({"thinking": MappingProxyType({k: thinking[k] for k in thinking})}) + thinking_mapping: Final = str_keyed_mapping_or_none(thinking) + if thinking_mapping is not None: + return MappingProxyType({"thinking": MappingProxyType(thinking_mapping)}) typ_only: Final = _thinking_type_value(thinking, thinking_values or ("enabled", "disabled")) if typ_only is None: - return MappingProxyType({}) + return _EMPTY return MappingProxyType({"thinking": MappingProxyType({"type": typ_only})}) case "enable_thinking": return MappingProxyType({"enable_thinking": _thinking_enabled(thinking)}) @@ -151,7 +161,7 @@ def _map_thinking_to_extra_body( {"chat_template_kwargs": MappingProxyType({"enable_thinking": _thinking_enabled(thinking)})} ) case _: - return MappingProxyType({}) + return _EMPTY def _map_effort_to_extra_body( @@ -165,7 +175,7 @@ def _map_effort_to_extra_body( return MappingProxyType({"reasoning_effort": clamped}) if not effort_values and send_via == _SEND_VIA_EXTRA_BODY and isinstance(effort, str): return MappingProxyType({"reasoning_effort": effort}) - return MappingProxyType({}) + return _EMPTY def translate_thinking_params( @@ -202,7 +212,7 @@ def translate_thinking_params( thinking_values=thinking_values, ) if thinking is not None and send_via == _SEND_VIA_EXTRA_BODY - else MappingProxyType({}) + else _EMPTY ) effort_patch: Final = ( _map_effort_to_extra_body( @@ -211,7 +221,7 @@ def translate_thinking_params( send_via=send_via, ) if effort is not None - else MappingProxyType({}) + else _EMPTY ) patch: Final = MappingProxyType({**thinking_patch, **effort_patch}) if not patch: diff --git a/litellm/utils.py b/litellm/utils.py index 0d71c65c3c9..6a5819ceeb5 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -296,6 +296,7 @@ from typing_extensions import assert_never from litellm import utils as litellm_utils from litellm.litellm_core_utils.thinking_param_translation import ( ThinkingParamsState, + str_keyed_mapping_or_none, thaw_mapping, translate_thinking_params, ) @@ -4329,14 +4330,13 @@ def remove_sensitive_keys_from_dict(d: dict) -> dict: def _translate_thinking_in_params( *, model_info: Mapping[str, object] | None, - passed_params: dict, # mutable-ok: get_optional_params hands over its legacy mutable params - non_default_params: dict, # mutable-ok: get_optional_params hands over its legacy mutable params -) -> tuple[dict, dict]: # mutable-ok: get_optional_params keeps mutating both copies downstream - existing_extra_body: Final = passed_params.get("extra_body") + passed_params: dict[str, object], # mutable-ok: get_optional_params hands over its legacy mutable params + non_default_params: dict[str, object], # mutable-ok: get_optional_params hands over its legacy mutable params +) -> tuple[dict[str, object], dict[str, object]]: # mutable-ok: get_optional_params keeps mutating both downstream state: Final = ThinkingParamsState( thinking=non_default_params.get("thinking"), reasoning_effort=non_default_params.get("reasoning_effort"), - extra_body=existing_extra_body if isinstance(existing_extra_body, Mapping) else MappingProxyType({}), + extra_body=str_keyed_mapping_or_none(passed_params.get("extra_body")) or MappingProxyType({}), ) translated: Final = translate_thinking_params(model_info=model_info, state=state) if translated is state: diff --git a/tests/unit/litellm_core_utils/test_thinking_param_translation.py b/tests/unit/litellm_core_utils/test_thinking_param_translation.py index a8cc0016bbd..956022ba7f9 100644 --- a/tests/unit/litellm_core_utils/test_thinking_param_translation.py +++ b/tests/unit/litellm_core_utils/test_thinking_param_translation.py @@ -210,6 +210,7 @@ def test_translate_moves_params_into_extra_body( pytest.param(_model_info(), {"type": "adaptive"}, None, id="thinking_type_not_allowed"), pytest.param(_model_info(), {"budget_tokens": 1024}, None, id="thinking_type_unreadable"), pytest.param(_model_info(), 1, None, id="thinking_type_unsupported_value"), + pytest.param(_model_info(), {1: "enabled"}, None, id="thinking_mapping_with_non_string_keys"), pytest.param( _model_info(thinking_param="thinking", thinking_values=[]), "adaptive", None, id="thinking_value_unmapped" ),