fix: drop inert model_info lookup from vLLM batch path

The vLLM batch path never forwards thinking or reasoning_effort to
get_optional_params, so the lookup could not change the request. It also
fell back to get_model_info, which completion() does not do. Non-vLLM
batch requests already go through completion() with the caller's
model_info

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
hx 2026-09-27 18:09:03 +08:00
parent 755efe014d
commit e18ff4b3b7

View file

@ -1,29 +1,13 @@
from collections.abc import Mapping
from concurrent.futures import FIRST_COMPLETED, ThreadPoolExecutor, wait
from typing import Final
import litellm
from litellm._logging import print_verbose
from litellm.utils import get_model_info, get_optional_params
from litellm.utils import get_optional_params
from ..llms.vllm.completion import handler as vllm_handler
def _model_info_for_batch(
*,
model: str,
custom_llm_provider: str | None,
kwargs: Mapping[str, object],
) -> Mapping[str, object] | None:
from_kwargs: Final = kwargs.get("model_info")
if isinstance(from_kwargs, Mapping):
return from_kwargs
try:
return get_model_info(model=model, custom_llm_provider=custom_llm_provider)
except Exception: # noqa: BLE001 # get_model_info raises Exception for unmapped models
return None
def batch_completion(
model: str,
# Optional OpenAI params: see https://platform.openai.com/docs/api-reference/chat/create
@ -95,13 +79,9 @@ def batch_completion(
frequency_penalty=frequency_penalty,
logit_bias=logit_bias,
user=user,
# params to identify the model
model=model,
custom_llm_provider=custom_llm_provider,
model_info=_model_info_for_batch(
model=model,
custom_llm_provider=custom_llm_provider,
kwargs=kwargs,
),
)
results = vllm_handler.batch_completions(
model=model,