From e18ff4b3b73be24313417a39b8cfeede7cca561f Mon Sep 17 00:00:00 2001 From: hx <1367557521@qq.com> Date: Sun, 27 Sep 2026 18:09:03 +0800 Subject: [PATCH] fix: drop inert model_info lookup from vLLM batch path The vLLM batch path never forwards thinking or reasoning_effort to get_optional_params, so the lookup could not change the request. It also fell back to get_model_info, which completion() does not do. Non-vLLM batch requests already go through completion() with the caller's model_info Co-authored-by: Cursor --- litellm/batch_completion/main.py | 24 ++---------------------- 1 file changed, 2 insertions(+), 22 deletions(-) diff --git a/litellm/batch_completion/main.py b/litellm/batch_completion/main.py index 7ec71dcd4bb..702dd194fda 100644 --- a/litellm/batch_completion/main.py +++ b/litellm/batch_completion/main.py @@ -1,29 +1,13 @@ -from collections.abc import Mapping from concurrent.futures import FIRST_COMPLETED, ThreadPoolExecutor, wait from typing import Final import litellm from litellm._logging import print_verbose -from litellm.utils import get_model_info, get_optional_params +from litellm.utils import get_optional_params from ..llms.vllm.completion import handler as vllm_handler -def _model_info_for_batch( - *, - model: str, - custom_llm_provider: str | None, - kwargs: Mapping[str, object], -) -> Mapping[str, object] | None: - from_kwargs: Final = kwargs.get("model_info") - if isinstance(from_kwargs, Mapping): - return from_kwargs - try: - return get_model_info(model=model, custom_llm_provider=custom_llm_provider) - except Exception: # noqa: BLE001 # get_model_info raises Exception for unmapped models - return None - - def batch_completion( model: str, # Optional OpenAI params: see https://platform.openai.com/docs/api-reference/chat/create @@ -95,13 +79,9 @@ def batch_completion( frequency_penalty=frequency_penalty, logit_bias=logit_bias, user=user, + # params to identify the model model=model, custom_llm_provider=custom_llm_provider, - model_info=_model_info_for_batch( - model=model, - custom_llm_provider=custom_llm_provider, - kwargs=kwargs, - ), ) results = vllm_handler.batch_completions( model=model,