Merge branch 'BerriAI:litellm_internal_staging' into feat/fireworks-think-tag-extraction

This commit is contained in:
Anurag Rao 2026-04-23 20:45:05 +05:30 committed by GitHub
commit 37bb6c1156
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
33 changed files with 2248 additions and 189 deletions

View file

@ -2061,7 +2061,7 @@ assert isinstance(
## Media Resolution Control (Images & Videos)
For Gemini 3+ models, LiteLLM supports per-part media resolution control using OpenAI's `detail` parameter. This allows you to specify different resolution levels for individual images and videos in your request, whether using `image_url` or `file` content types.
LiteLLM supports per-part media resolution control using OpenAI's `detail` parameter for all Gemini models. This allows you to specify different resolution levels for individual images and videos in your request, whether using `image_url` or `file` content types.
**Supported `detail` values:**
- `"low"` - Maps to `media_resolution: "low"` (280 tokens for images, 70 tokens per frame for videos)
@ -2146,12 +2146,12 @@ response = completion(
</Tabs>
:::info
**Per-Part Resolution:** Each image or video in your request can have its own `detail` setting, allowing mixed-resolution requests (e.g., a high-res chart alongside a low-res icon). This feature works with both `image_url` and `file` content types, and is only available for Gemini 3+ models.
**Per-Part Resolution:** Each image or video in your request can have its own `detail` setting, allowing mixed-resolution requests (e.g., a high-res chart alongside a low-res icon). This feature works with both `image_url` and `file` content types across all Gemini models.
:::
## Video Metadata Control
For Gemini 3+ models, LiteLLM supports fine-grained video processing control through the `video_metadata` field. This allows you to specify frame extraction rates and time ranges for video analysis.
LiteLLM supports fine-grained video processing control through the `video_metadata` field for all Gemini models (1.x, 2.x, 3+). This allows you to specify frame extraction rates and time ranges for video analysis.
**Supported `video_metadata` parameters:**
@ -2168,8 +2168,11 @@ For Gemini 3+ models, LiteLLM supports fine-grained video processing control thr
- `fps` remains unchanged
:::
:::tip
Video clipping (`start_offset`/`end_offset`) and frame rate control (`fps`) are supported by all Gemini models, but analysis quality is significantly higher with the **Gemini 2.5 series** (e.g., `gemini-2.5-flash`, `gemini-2.5-pro`).
:::
:::warning
- **Gemini 3+ Only:** This feature is only available for Gemini 3.0 and newer models
- **Video Files Recommended:** While `video_metadata` is designed for video files, error handling for other media types is delegated to the Vertex AI API
- **File Formats Supported:** Works with `gs://`, `https://`, and base64-encoded video files
:::

View file

@ -410,6 +410,7 @@ def image_generation( # noqa: PLR0915
litellm.LlmProviders.RUNWAYML,
litellm.LlmProviders.VERTEX_AI,
litellm.LlmProviders.OPENROUTER,
litellm.LlmProviders.DASHSCOPE,
):
if image_generation_config is None:
raise ValueError(

View file

@ -452,7 +452,14 @@ def update_messages_with_model_file_ids(
for c in content:
if c["type"] == "file":
file_object = cast(ChatCompletionFileObject, c)
file_object_file_field = file_object["file"]
file_object_file_field = file_object.get("file")
if not isinstance(file_object_file_field, dict):
# Content block has `type: "file"` but not the
# OpenAI Chat Completions shape (e.g. a LangChain
# v1 standardized file block, or a provider-native
# shape that also uses `type: "file"`). Nothing to
# remap here, so skip instead of crashing.
continue
file_id = file_object_file_field.get("file_id")
format = file_object_file_field.get(
"format", get_format_from_file_id(file_id)
@ -1060,7 +1067,12 @@ def get_file_ids_from_messages(messages: List[AllMessageValues]) -> List[str]:
for c in content:
if c["type"] == "file":
file_object = cast(ChatCompletionFileObject, c)
file_object_file_field = file_object["file"]
file_object_file_field = file_object.get("file")
if not isinstance(file_object_file_field, dict):
# Content block has `type: "file"` but not the
# OpenAI Chat Completions shape. No file_id to
# extract, so skip instead of raising KeyError.
continue
file_id = file_object_file_field.get("file_id")
if file_id:
file_ids.append(file_id)

View file

@ -106,6 +106,44 @@ class LiteLLMMessagesToCompletionTransformationHandler:
updated_reasoning_effort["summary"] = effective_summary
completion_kwargs["reasoning_effort"] = updated_reasoning_effort
@staticmethod
def _normalize_reasoning_effort(
completion_kwargs: Dict[str, Any],
) -> None:
"""
Normalize reasoning_effort values based on target model capabilities.
Handles both string ("max") and dict ({"effort": "max", "summary": ...})
formats. Uses model registry to check supports_xhigh/supports_minimal.
"""
from litellm.llms.anthropic.experimental_pass_through.utils import (
normalize_reasoning_effort_value,
)
reasoning_effort = completion_kwargs.get("reasoning_effort")
if reasoning_effort is None:
return
model = cast(str, completion_kwargs.get("model", ""))
custom_llm_provider = completion_kwargs.get("custom_llm_provider")
if isinstance(reasoning_effort, str):
normalized = normalize_reasoning_effort_value(
reasoning_effort, model=model, custom_llm_provider=custom_llm_provider
)
if normalized != reasoning_effort:
completion_kwargs["reasoning_effort"] = normalized
elif isinstance(reasoning_effort, dict) and "effort" in reasoning_effort:
effort = reasoning_effort["effort"]
normalized = normalize_reasoning_effort_value(
effort, model=model, custom_llm_provider=custom_llm_provider
)
if normalized != effort:
completion_kwargs["reasoning_effort"] = {
**reasoning_effort,
"effort": normalized,
}
@staticmethod
def _prepare_completion_kwargs(
*,
@ -163,6 +201,12 @@ class LiteLLMMessagesToCompletionTransformationHandler:
if output_format:
request_data["output_format"] = output_format
# Extract output_config from extra_kwargs so the translator can use it
# (e.g. output_config.effort for adaptive thinking → reasoning_effort)
extra_kwargs = extra_kwargs or {}
if "output_config" in extra_kwargs:
request_data["output_config"] = extra_kwargs["output_config"]
(
openai_request,
tool_name_mapping,
@ -202,6 +246,14 @@ class LiteLLMMessagesToCompletionTransformationHandler:
):
completion_kwargs[key] = value
# Normalize reasoning_effort based on model capabilities
# (e.g. "max" → "xhigh"/"high", "minimal" → "low" if unsupported)
# Must run BEFORE _route_openai_thinking, which prepends "responses/"
# to the model name and would break get_model_info() lookups.
LiteLLMMessagesToCompletionTransformationHandler._normalize_reasoning_effort(
completion_kwargs
)
LiteLLMMessagesToCompletionTransformationHandler._route_openai_thinking_to_responses_api_if_needed(
completion_kwargs,
thinking=thinking,

View file

@ -317,6 +317,7 @@ class LiteLLMAnthropicMessagesAdapter:
"tools",
"thinking",
"output_format",
"output_config",
]
def _is_web_search_tool(self, tool: Dict[str, Any]) -> bool:
@ -694,6 +695,11 @@ class LiteLLMAnthropicMessagesAdapter:
return "low"
else:
return "minimal"
elif thinking_type == "adaptive":
# Adaptive thinking: effort is controlled by output_config.effort,
# not budget_tokens. Return a default; caller should override with
# output_config.effort when available.
return "medium"
return None
@ -776,6 +782,8 @@ class LiteLLMAnthropicMessagesAdapter:
return ChatCompletionToolChoiceObjectParam(
type="function", function=tc_function_param
)
elif tool_choice["type"] == "none":
return "none"
else:
raise ValueError(
"Incompatible tool choice param submitted - {}".format(tool_choice)
@ -1041,6 +1049,12 @@ class LiteLLMAnthropicMessagesAdapter:
if not reasoning_effort:
return
# For adaptive thinking, override with output_config.effort if available
if isinstance(thinking, dict) and thinking.get("type") == "adaptive":
output_config = anthropic_message_request.get("output_config")
if isinstance(output_config, dict) and output_config.get("effort"):
reasoning_effort = output_config["effort"]
summary = thinking.get("summary") if isinstance(thinking, dict) else None
auto_summary = is_reasoning_auto_summary_enabled()
if summary:

View file

@ -24,6 +24,8 @@ from litellm.types.llms.anthropic_messages.anthropic_response import (
from litellm.types.router import GenericLiteLLMParams
from litellm.utils import ProviderConfigManager, client
from ..utils import is_reasoning_auto_summary_enabled
from ..adapters.handler import LiteLLMMessagesToCompletionTransformationHandler
from ..responses_adapters.handler import LiteLLMMessagesToResponsesAPIHandler
from .interceptors import get_messages_interceptors
@ -441,6 +443,17 @@ def anthropic_messages_handler(
params=local_vars
)
)
if is_reasoning_auto_summary_enabled():
thinking_param = anthropic_messages_optional_request_params.get("thinking")
if (
isinstance(thinking_param, dict)
and thinking_param.get("type") != "disabled"
):
anthropic_messages_optional_request_params["thinking"] = {
**thinking_param,
"display": "summarized",
}
return base_llm_http_handler.anthropic_messages_handler(
model=model,
messages=messages,

View file

@ -72,6 +72,23 @@ def _build_responses_kwargs(
anthropic_request = AnthropicMessagesRequest(**request_data) # type: ignore[typeddict-item]
responses_kwargs = _ADAPTER.translate_request(anthropic_request)
# Normalize reasoning effort based on model capabilities
# (e.g. "max" → "xhigh"/"high", "minimal" → "low" if unsupported)
reasoning = responses_kwargs.get("reasoning")
if isinstance(reasoning, dict) and "effort" in reasoning:
from litellm.llms.anthropic.experimental_pass_through.utils import (
normalize_reasoning_effort_value,
)
effort = reasoning["effort"]
normalized = normalize_reasoning_effort_value(
effort,
model=model,
custom_llm_provider=(extra_kwargs or {}).get("custom_llm_provider"),
)
if normalized != effort:
responses_kwargs["reasoning"] = {**reasoning, "effort": normalized}
if stream:
responses_kwargs["stream"] = True

View file

@ -251,25 +251,41 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
@staticmethod
def translate_thinking_to_reasoning(
thinking: Dict[str, Any]
thinking: Dict[str, Any],
output_config: Optional[Dict[str, Any]] = None,
) -> Optional[Dict[str, Any]]:
"""
Convert Anthropic thinking param to Responses API reasoning param.
thinking.budget_tokens maps to reasoning effort:
>= 10000 -> high, >= 5000 -> medium, >= 2000 -> low, < 2000 -> minimal
For adaptive thinking, uses output_config.effort if available,
otherwise defaults to medium.
"""
if not isinstance(thinking, dict) or thinking.get("type") != "enabled":
if not isinstance(thinking, dict):
return None
budget = thinking.get("budget_tokens", 0)
if budget >= 10000:
effort = "high"
elif budget >= 5000:
thinking_type = thinking.get("type")
if thinking_type == "adaptive":
# Use output_config.effort if available
effort = "medium"
elif budget >= 2000:
effort = "low"
if isinstance(output_config, dict) and output_config.get("effort"):
effort = output_config["effort"]
elif thinking_type == "enabled":
budget = thinking.get("budget_tokens", 0)
if budget >= 10000:
effort = "high"
elif budget >= 5000:
effort = "medium"
elif budget >= 2000:
effort = "low"
else:
effort = "minimal"
else:
effort = "minimal"
return None
auto_summary = is_reasoning_auto_summary_enabled()
result: Dict[str, Any] = {"effort": effort}
summary = thinking.get("summary")
@ -346,7 +362,11 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
# thinking -> reasoning
thinking = anthropic_request.get("thinking")
if isinstance(thinking, dict):
reasoning = self.translate_thinking_to_reasoning(thinking)
output_config = anthropic_request.get("output_config")
reasoning = self.translate_thinking_to_reasoning(
thinking,
output_config=cast(Optional[Dict[str, Any]], output_config),
)
if reasoning:
responses_kwargs["reasoning"] = reasoning

View file

@ -1,6 +1,8 @@
import os
from typing import Optional
import litellm
from litellm.types.utils import ModelInfo
def is_reasoning_auto_summary_enabled() -> bool:
@ -9,3 +11,47 @@ def is_reasoning_auto_summary_enabled() -> bool:
litellm.reasoning_auto_summary
or os.getenv("LITELLM_REASONING_AUTO_SUMMARY", "false").lower() == "true"
)
def normalize_reasoning_effort_value(
effort: str,
model: str,
custom_llm_provider: Optional[str] = None,
) -> str:
"""
Normalize a reasoning effort value based on model capabilities.
Degradation chains:
- "max" max / xhigh / high
- "xhigh" xhigh / high
- "minimal" minimal / low
- other values pass through unchanged
"""
if effort not in ("max", "xhigh", "minimal"):
return effort
from litellm.utils import get_model_info
model_info: Optional[ModelInfo] = None
try:
model_info = get_model_info(
model=model, custom_llm_provider=custom_llm_provider
)
except Exception:
model_info = None
if effort == "max":
if model_info and model_info.get("supports_max_reasoning_effort"):
return "max"
if model_info and model_info.get("supports_xhigh_reasoning_effort"):
return "xhigh"
return "high"
elif effort == "xhigh":
if model_info and model_info.get("supports_xhigh_reasoning_effort"):
return "xhigh"
return "high"
elif effort == "minimal":
if model_info and model_info.get("supports_minimal_reasoning_effort"):
return "minimal"
return "low"
return "medium"

View file

@ -40,9 +40,22 @@ class AzureOpenAIGPT5Config(AzureOpenAIConfig, OpenAIGPT5Config):
Accepts both explicit gpt-5 model names and the ``gpt5_series/`` prefix
used for manual routing.
"""
# gpt-5-chat* is a chat model and shouldn't go through GPT-5 reasoning restrictions.
# The gpt-5-chat* family (gpt-5-chat, gpt-5-chat-latest, gpt-5-chat-2025-08-07,
# …) are regular chat models: they support temperature and tool_choice but NOT
# reasoning_effort. They must NOT be routed through the GPT-5 reasoning path.
#
# Versioned chat models such as gpt-5.3-chat and gpt-5.1-chat ARE reasoning
# models and must stay on the GPT-5 path. The distinguishing feature is that
# the gpt-5-chat family has a literal "-chat" immediately after "gpt-5"
# (i.e. "gpt-5-chat…"), while versioned chat models interpose a minor version
# number (i.e. "gpt-5.<digit>-chat").
#
# Using a startswith("gpt-5-chat") prefix check on the normalized name (rather
# than a substring check) makes this boundary explicit and avoids any ambiguity
# if future model names coincidentally contain "gpt-5-chat" as an interior run.
_normalized = model.split("/")[-1] # strip provider prefix, e.g. "azure/"
return (
"gpt-5" in model and "gpt-5-chat" not in model
"gpt-5" in model and not _normalized.startswith("gpt-5-chat")
) or "gpt5_series" in model
def get_supported_openai_params(self, model: str) -> List[str]:

View file

@ -0,0 +1,11 @@
from litellm.llms.base_llm.image_generation.transformation import (
BaseImageGenerationConfig,
)
from .transformation import DashScopeImageGenerationConfig
__all__ = ["DashScopeImageGenerationConfig"]
def get_dashscope_image_generation_config(model: str) -> BaseImageGenerationConfig:
return DashScopeImageGenerationConfig()

View file

@ -0,0 +1,204 @@
"""
DashScope Image Generation Configuration
Handles transformation between OpenAI-compatible format and DashScope multimodal-generation API.
API endpoint: POST https://dashscope-intl.aliyuncs.com/api/v1/services/aigc/multimodal-generation/generation
Request format:
{
"model": "qwen-image-2.0-pro",
"input": {
"messages": [{"role": "user", "content": [{"text": "<prompt>"}]}]
},
"parameters": {"size": "1024*1024", ...}
}
Response format:
{
"output": {
"choices": [{"message": {"content": [{"image": "<url>"}]}}]
},
"usage": {"input_tokens": 0, "output_tokens": 0, "width": 1024, "height": 1024, "image_count": 1}
}
"""
from typing import TYPE_CHECKING, Any, List, Optional
import httpx
from litellm.llms.base_llm.image_generation.transformation import (
BaseImageGenerationConfig,
)
from litellm.secret_managers.main import get_secret_str
from litellm.types.llms.openai import (
AllMessageValues,
OpenAIImageGenerationOptionalParams,
)
from litellm.types.utils import ImageObject, ImageResponse
if TYPE_CHECKING:
from litellm.litellm_core_utils.litellm_logging import Logging as _LiteLLMLoggingObj
LiteLLMLoggingObj = _LiteLLMLoggingObj
else:
LiteLLMLoggingObj = Any
DEFAULT_API_BASE = "https://dashscope-intl.aliyuncs.com/api/v1/services/aigc/multimodal-generation/generation"
# Maps OpenAI size strings (WxH) to DashScope size strings (W*H)
OPENAI_TO_DASHSCOPE_SIZE: dict = {
"256x256": "256*256",
"512x512": "512*512",
"1024x1024": "1024*1024",
"1792x1024": "1792*1024",
"1024x1792": "1024*1792",
"2048x2048": "2048*2048",
}
class DashScopeImageGenerationConfig(BaseImageGenerationConfig):
"""
Configuration for DashScope image generation (qwen-image-2.0, qwen-image-2.0-pro).
"""
def get_supported_openai_params(
self, model: str
) -> List[OpenAIImageGenerationOptionalParams]:
return ["n", "size"]
def map_openai_params(
self,
non_default_params: dict,
optional_params: dict,
model: str,
drop_params: bool,
) -> dict:
supported_params = self.get_supported_openai_params(model)
mapped: dict = {}
for k, v in non_default_params.items():
if k in optional_params:
continue
if k not in supported_params:
continue
if k == "size":
# Convert "WxH" → "W*H"
mapped["size"] = OPENAI_TO_DASHSCOPE_SIZE.get(v, v.replace("x", "*"))
elif k == "n":
mapped["image_count"] = v
return mapped
def get_complete_url(
self,
api_base: Optional[str],
api_key: Optional[str],
model: str,
optional_params: dict,
litellm_params: dict,
stream: Optional[bool] = None,
) -> str:
return (
api_base or get_secret_str("DASHSCOPE_API_BASE_IMAGE") or DEFAULT_API_BASE
)
def validate_environment(
self,
headers: dict,
model: str,
messages: List[AllMessageValues],
optional_params: dict,
litellm_params: dict,
api_key: Optional[str] = None,
api_base: Optional[str] = None,
) -> dict:
final_api_key = api_key or get_secret_str("DASHSCOPE_API_KEY")
if not final_api_key:
raise ValueError("DASHSCOPE_API_KEY is not set")
headers["Authorization"] = f"Bearer {final_api_key}"
headers["Content-Type"] = "application/json"
return headers
def transform_image_generation_request(
self,
model: str,
prompt: str,
optional_params: dict,
litellm_params: dict,
headers: dict,
) -> dict:
"""
Transform OpenAI-style image generation request to DashScope multimodal-generation format.
"""
parameters: dict = {}
for k, v in optional_params.items():
parameters[k] = v
return {
"model": model,
"input": {
"messages": [
{
"role": "user",
"content": [{"text": prompt}],
}
]
},
"parameters": parameters,
}
def transform_image_generation_response(
self,
model: str,
raw_response: httpx.Response,
model_response: ImageResponse,
logging_obj: LiteLLMLoggingObj,
request_data: dict,
optional_params: dict,
litellm_params: dict,
encoding: Any,
api_key: Optional[str] = None,
json_mode: Optional[bool] = None,
) -> ImageResponse:
"""
Transform DashScope response to litellm ImageResponse.
DashScope response: output.choices[0].message.content[0].image
OpenAI response: data[0].url
"""
if raw_response.status_code != 200:
raise self.get_error_class(
error_message=raw_response.text,
status_code=raw_response.status_code,
headers=raw_response.headers,
)
try:
response_data = raw_response.json()
except Exception as e:
raise self.get_error_class(
error_message=f"Failed to parse DashScope image generation response: {e}",
status_code=raw_response.status_code,
headers=raw_response.headers,
)
# DashScope can return API-level errors in a 200 response body.
# Example: {"code": "InvalidParameter", "message": "Size not supported"}
if "code" in response_data and "output" not in response_data:
raise self.get_error_class(
error_message=str(response_data.get("message", response_data)),
status_code=raw_response.status_code,
headers=raw_response.headers,
)
if not model_response.data:
model_response.data = []
choices = response_data.get("output", {}).get("choices", [])
for choice in choices:
content_list = choice.get("message", {}).get("content", [])
for content_item in content_list:
image_url = content_item.get("image")
if image_url:
model_response.data.append(ImageObject(url=image_url))
return model_response

View file

@ -53,9 +53,21 @@ class OpenAIGPT5Config(OpenAIGPTConfig):
@classmethod
def is_model_gpt_5_model(cls, model: str) -> bool:
# gpt-5-chat* behaves like a regular chat model (supports temperature, etc.)
# Don't route it through GPT-5 reasoning-specific parameter restrictions.
return "gpt-5" in model and "gpt-5-chat" not in model
# The gpt-5-chat* family (gpt-5-chat, gpt-5-chat-latest, gpt-5-chat-2025-08-07,
# …) are regular chat models: they support temperature and tool_choice but NOT
# reasoning_effort. They must NOT be routed through the GPT-5 reasoning path.
#
# Versioned chat models such as gpt-5.3-chat and gpt-5.1-chat ARE reasoning
# models and must stay on the GPT-5 path. The distinguishing feature is that
# the gpt-5-chat family has a literal "-chat" immediately after "gpt-5"
# (i.e. "gpt-5-chat…"), while versioned chat models interpose a minor version
# number (i.e. "gpt-5.<digit>-chat").
#
# Using a startswith("gpt-5-chat") prefix check on the normalized name (rather
# than a substring check) makes this boundary explicit and avoids any ambiguity
# if future model names coincidentally contain "gpt-5-chat" as an interior run.
_normalized = model.split("/")[-1] # strip provider prefix, e.g. "openai/"
return "gpt-5" in model and not _normalized.startswith("gpt-5-chat")
@classmethod
def is_model_gpt_5_search_model(cls, model: str) -> bool:

View file

@ -8,9 +8,8 @@ More information on our website: https://endpoints.ai.cloud.ovh.net
from typing import Optional, Union, List
import httpx
from litellm.utils import ModelResponseStream, _get_model_info_helper
from litellm.utils import ModelResponseStream
from litellm.llms.openai.chat.gpt_transformation import OpenAIGPTConfig
from litellm._logging import verbose_logger
from litellm.llms.ovhcloud.utils import OVHCloudException
from litellm.llms.base_llm.base_model_iterator import BaseModelResponseIterator
from litellm.llms.base_llm.chat.transformation import BaseLLMException
@ -22,34 +21,6 @@ class OVHCloudChatConfig(OpenAIGPTConfig):
def custom_llm_provider(self) -> Optional[str]:
return "ovhcloud"
def get_supported_openai_params(self, model: str) -> list:
"""
Details about function calling support can be found here:
https://help.ovhcloud.com/csm/en-gb-public-cloud-ai-endpoints-function-calling?id=kb_article_view&sysparm_article=KB0071907
"""
supports_function_calling: Optional[bool] = None
try:
model_info = _get_model_info_helper(model, custom_llm_provider="ovhcloud")
supports_function_calling = model_info.get(
"supports_function_calling", None
)
if supports_function_calling is None:
supports_function_calling = False
except Exception as e:
verbose_logger.debug(f"Error getting supported OpenAI params: {e}")
supports_function_calling = False
optional_params = super().get_supported_openai_params(model)
if supports_function_calling is not True:
verbose_logger.debug(
"You can see our models supporting function_calling in our catalog: https://endpoints.ai.cloud.ovh.net/catalog "
)
optional_params.remove("tools")
optional_params.remove("tool_choice")
optional_params.remove("function_call")
optional_params.remove("response_format")
return optional_params
def get_complete_url(
self,
api_base: Optional[str],

View file

@ -132,26 +132,28 @@ def _extract_max_media_resolution_from_messages(
return max_resolution
def _apply_gemini_3_metadata(
def _apply_gemini_metadata(
part: PartType,
model: Optional[str],
media_resolution_enum: Optional[Dict[str, str]],
video_metadata: Optional[Dict[str, Any]],
) -> PartType:
"""
Apply the unique media_resolution and video_metadata parameters of Gemini 3+
Apply media_resolution and video_metadata parameters to a Gemini part.
- Per-part media_resolution: Gemini 3+ only (2.x uses generation_config global).
- video_metadata (fps, startOffset, endOffset): all Gemini models (1.x, 2.x, 3+).
"""
if model is None:
return part
from .vertex_and_google_ai_studio_gemini import VertexGeminiConfig
if not VertexGeminiConfig._is_gemini_3_or_newer(model):
return part
part_dict = dict(part)
if media_resolution_enum is not None:
if media_resolution_enum is not None and VertexGeminiConfig._is_gemini_3_or_newer(
model
):
part_dict["media_resolution"] = media_resolution_enum
if video_metadata is not None:
@ -206,7 +208,7 @@ def _process_gemini_media(
mime_type = format
file_data = FileDataType(mime_type=mime_type, file_uri=image_url)
part: PartType = {"file_data": file_data}
return _apply_gemini_3_metadata(
return _apply_gemini_metadata(
part, model, media_resolution_enum, video_metadata
)
elif (
@ -216,14 +218,14 @@ def _process_gemini_media(
):
file_data = FileDataType(mime_type=image_type, file_uri=image_url)
part = {"file_data": file_data}
return _apply_gemini_3_metadata(
return _apply_gemini_metadata(
part, model, media_resolution_enum, video_metadata
)
elif "http://" in image_url or "https://" in image_url or "base64" in image_url:
image = convert_to_anthropic_image_obj(image_url, format=format)
_blob: BlobType = {"data": image["data"], "mime_type": image["media_type"]}
part = {"inline_data": cast(BlobType, _blob)}
return _apply_gemini_3_metadata(
return _apply_gemini_metadata(
part, model, media_resolution_enum, video_metadata
)
raise Exception("Invalid image received - {}".format(image_url))
@ -733,9 +735,9 @@ def _transform_request_body( # noqa: PLR0915
**filtered_params
)
# For Gemini 2.x models, add media_resolution to generation_config (global)
# Gemini 3+ supports per-part media_resolution, but 2.x only supports global
# Gemini 1.x does not support mediaResolution at all
# For Gemini 2.x models, also add media_resolution to generation_config (global)
# as a fallback, since some 2.x versions may not support per-part media_resolution.
# Gemini 1.x does not support mediaResolution at all.
if "gemini-2" in model:
max_media_resolution = _extract_max_media_resolution_from_messages(messages)
if max_media_resolution:

View file

@ -1006,7 +1006,8 @@
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true,
"supports_max_reasoning_effort": true
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"global.anthropic.claude-opus-4-6-v1": {
"cache_creation_input_token_cost": 6.25e-06,
@ -1034,7 +1035,8 @@
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true,
"supports_max_reasoning_effort": true
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"us.anthropic.claude-opus-4-6-v1": {
"cache_creation_input_token_cost": 6.875e-06,
@ -1062,7 +1064,8 @@
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true,
"supports_max_reasoning_effort": true
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"eu.anthropic.claude-opus-4-6-v1": {
"cache_creation_input_token_cost": 6.875e-06,
@ -1090,7 +1093,8 @@
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true,
"supports_max_reasoning_effort": true
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"au.anthropic.claude-opus-4-6-v1": {
"cache_creation_input_token_cost": 6.875e-06,
@ -1118,7 +1122,8 @@
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true,
"supports_max_reasoning_effort": true
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"anthropic.claude-opus-4-7": {
"cache_creation_input_token_cost": 6.25e-06,
@ -1146,7 +1151,9 @@
"supports_vision": true,
"supports_xhigh_reasoning_effort": true,
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
"supports_native_structured_output": true,
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"global.anthropic.claude-opus-4-7": {
"cache_creation_input_token_cost": 6.25e-06,
@ -1174,7 +1181,9 @@
"supports_vision": true,
"supports_xhigh_reasoning_effort": true,
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
"supports_native_structured_output": true,
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"us.anthropic.claude-opus-4-7": {
"cache_creation_input_token_cost": 6.875e-06,
@ -1202,7 +1211,9 @@
"supports_vision": true,
"supports_xhigh_reasoning_effort": true,
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
"supports_native_structured_output": true,
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"eu.anthropic.claude-opus-4-7": {
"cache_creation_input_token_cost": 6.875e-06,
@ -1230,7 +1241,9 @@
"supports_vision": true,
"supports_xhigh_reasoning_effort": true,
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
"supports_native_structured_output": true,
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"au.anthropic.claude-opus-4-7": {
"cache_creation_input_token_cost": 6.875e-06,
@ -1258,7 +1271,9 @@
"supports_vision": true,
"supports_xhigh_reasoning_effort": true,
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
"supports_native_structured_output": true,
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"anthropic.claude-sonnet-4-6": {
"cache_creation_input_token_cost": 3.75e-06,
@ -1285,7 +1300,8 @@
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
"supports_native_structured_output": true,
"supports_minimal_reasoning_effort": true
},
"global.anthropic.claude-sonnet-4-6": {
"cache_creation_input_token_cost": 3.75e-06,
@ -1312,7 +1328,8 @@
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
"supports_native_structured_output": true,
"supports_minimal_reasoning_effort": true
},
"us.anthropic.claude-sonnet-4-6": {
"cache_creation_input_token_cost": 4.125e-06,
@ -1339,7 +1356,8 @@
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
"supports_native_structured_output": true,
"supports_minimal_reasoning_effort": true
},
"eu.anthropic.claude-sonnet-4-6": {
"cache_creation_input_token_cost": 4.125e-06,
@ -1366,7 +1384,8 @@
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
"supports_native_structured_output": true,
"supports_minimal_reasoning_effort": true
},
"au.anthropic.claude-sonnet-4-6": {
"cache_creation_input_token_cost": 4.125e-06,
@ -1393,7 +1412,8 @@
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
"supports_native_structured_output": true,
"supports_minimal_reasoning_effort": true
},
"anthropic.claude-sonnet-4-20250514-v1:0": {
"cache_creation_input_token_cost": 3.75e-06,
@ -1911,7 +1931,8 @@
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 159,
"supports_max_reasoning_effort": true
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"azure_ai/claude-opus-4-7": {
"input_cost_per_token": 5e-06,
@ -1939,7 +1960,9 @@
"supports_tool_choice": true,
"supports_vision": true,
"supports_xhigh_reasoning_effort": true,
"tool_use_system_prompt_tokens": 159
"tool_use_system_prompt_tokens": 159,
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"azure_ai/claude-opus-4-1": {
"cache_creation_input_token_cost": 1.875e-05,
@ -2003,7 +2026,8 @@
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
"tool_use_system_prompt_tokens": 346,
"supports_minimal_reasoning_effort": true
},
"azure/computer-use-preview": {
"input_cost_per_token": 3e-06,
@ -8909,7 +8933,8 @@
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
"tool_use_system_prompt_tokens": 346,
"supports_minimal_reasoning_effort": true
},
"claude-sonnet-4-5-20250929-v1:0": {
"cache_creation_input_token_cost": 3.75e-06,
@ -9103,7 +9128,8 @@
"us": 1.1,
"fast": 6.0
},
"supports_max_reasoning_effort": true
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"claude-opus-4-6-20260205": {
"cache_creation_input_token_cost": 6.25e-06,
@ -9135,7 +9161,8 @@
"us": 1.1,
"fast": 6.0
},
"supports_max_reasoning_effort": true
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"claude-opus-4-7": {
"cache_creation_input_token_cost": 6.25e-06,
@ -9167,7 +9194,9 @@
"provider_specific_entry": {
"us": 1.1,
"fast": 6.0
}
},
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"claude-opus-4-7-20260416": {
"cache_creation_input_token_cost": 6.25e-06,
@ -9199,7 +9228,9 @@
"provider_specific_entry": {
"us": 1.1,
"fast": 6.0
}
},
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"claude-sonnet-4-20250514": {
"deprecation_date": "2026-05-14",
@ -10352,6 +10383,22 @@
"supports_reasoning": true,
"supports_tool_choice": true
},
"dashscope/qwen-image-2.0": {
"litellm_provider": "dashscope",
"mode": "image_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/models",
"supported_endpoints": [
"/v1/images/generations"
]
},
"dashscope/qwen-image-2.0-pro": {
"litellm_provider": "dashscope",
"mode": "image_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/models",
"supported_endpoints": [
"/v1/images/generations"
]
},
"databricks/databricks-bge-large-en": {
"input_cost_per_token": 1.0003e-07,
"input_dbu_cost_per_token": 1.429e-06,
@ -25068,7 +25115,8 @@
"supports_reasoning": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 159
"tool_use_system_prompt_tokens": 159,
"supports_minimal_reasoning_effort": true
},
"openrouter/anthropic/claude-opus-4.5": {
"cache_creation_input_token_cost": 6.25e-06,
@ -25106,7 +25154,8 @@
"supports_reasoning": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
"tool_use_system_prompt_tokens": 346,
"supports_minimal_reasoning_effort": true
},
"openrouter/anthropic/claude-sonnet-4.5": {
"input_cost_per_image": 0.0048,
@ -30134,7 +30183,8 @@
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true
"supports_vision": true,
"supports_minimal_reasoning_effort": true
},
"vercel_ai_gateway/anthropic/claude-sonnet-4": {
"cache_creation_input_token_cost": 3.75e-06,
@ -31361,7 +31411,8 @@
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
"supports_max_reasoning_effort": true
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"vertex_ai/claude-opus-4-6@default": {
"cache_creation_input_token_cost": 6.25e-06,
@ -31388,7 +31439,8 @@
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
"supports_max_reasoning_effort": true
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"vertex_ai/claude-opus-4-7": {
"cache_creation_input_token_cost": 6.25e-06,
@ -31415,7 +31467,9 @@
"supports_tool_choice": true,
"supports_vision": true,
"supports_xhigh_reasoning_effort": true,
"tool_use_system_prompt_tokens": 346
"tool_use_system_prompt_tokens": 346,
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"vertex_ai/claude-opus-4-7@default": {
"cache_creation_input_token_cost": 6.25e-06,
@ -31442,7 +31496,9 @@
"supports_tool_choice": true,
"supports_vision": true,
"supports_xhigh_reasoning_effort": true,
"tool_use_system_prompt_tokens": 346
"tool_use_system_prompt_tokens": 346,
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"vertex_ai/claude-sonnet-4-5": {
"cache_creation_input_token_cost": 3.75e-06,
@ -31494,7 +31550,8 @@
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
"search_context_size_medium": 0.01
}
},
"supports_minimal_reasoning_effort": true
},
"vertex_ai/claude-sonnet-4-5@20250929": {
"cache_creation_input_token_cost": 3.75e-06,
@ -38361,7 +38418,8 @@
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
"search_context_size_medium": 0.01
}
},
"supports_minimal_reasoning_effort": true
},
"duckduckgo/search": {
"litellm_provider": "duckduckgo",

View file

@ -7,6 +7,7 @@ Filters MCP tools semantically for /chat/completions and /responses endpoints.
from typing import TYPE_CHECKING, Any, Dict, List, Optional
from litellm._logging import verbose_logger
from litellm.proxy._experimental.mcp_server.utils import MCP_TOOL_PREFIX_SEPARATOR
if TYPE_CHECKING:
from semantic_router.routers import SemanticRouter
@ -214,20 +215,89 @@ class SemanticMCPToolFilter:
return []
@staticmethod
def _name_matches_canonical(client_name: str, canonical: str) -> bool:
"""
Return True if a client-side tool name refers to the given canonical
MCP tool name.
MCP clients (e.g. opencode) commonly wrap the proxy's canonical tool
name with an additive namespace prefix of their own
(``<client_alias><sep><canonical>``). The prefix can use either a
dash or an underscore as separator regardless of what
``MCP_TOOL_PREFIX_SEPARATOR`` is set to on the proxy, because the
client doesn't know the proxy's separator.
The match is anchored: ``canonical`` must form the complete suffix
of ``client_name`` and be preceded by a separator character, so
``rain_gear`` does not match canonical ``ear``.
Suffix matching is additionally gated on ``canonical`` itself
containing ``MCP_TOOL_PREFIX_SEPARATOR``. Server-registered MCP
tools are always emitted as
``<server_name><MCP_TOOL_PREFIX_SEPARATOR><tool_name>`` (see
``add_server_prefix_to_name``), so a canonical without the
separator is not a namespaced MCP tool and falling back to
suffix matching would spuriously collide with unrelated local
user functions whose names end in the same characters.
"""
if client_name == canonical:
return True
if MCP_TOOL_PREFIX_SEPARATOR not in canonical:
return False
if len(client_name) <= len(canonical):
return False
if not client_name.endswith(canonical):
return False
separator = client_name[-len(canonical) - 1]
return separator in ("_", "-")
def _get_tools_by_names(
self, tool_names: List[str], available_tools: List[Any]
) -> List[Any]:
"""Get tools from available_tools by their names, preserving order."""
# Match tools from available_tools (preserves format - dict or MCPTool)
matched_tools = []
for tool in available_tools:
tool_name, _ = self._extract_tool_info(tool)
if tool_name in tool_names:
matched_tools.append(tool)
"""
Get tools from available_tools by their names, preserving the
semantic router's ordering.
# Reorder to match semantic router's ordering
tool_map = {self._extract_tool_info(t)[0]: t for t in matched_tools}
return [tool_map[name] for name in tool_names if name in tool_map]
Matching is tolerant of client-side namespace prefixes: if an
incoming tool arrived as ``<client_alias>_<canonical>`` while the
router returned ``<canonical>`` (see
``_name_matches_canonical``), that tool is still selected. The
returned tool object is the original from ``available_tools``, so
the client-facing name is preserved for tool-call round-trips.
"""
# Build an index of incoming tools by their client-facing name.
# Exact matches win over suffix matches when both are present, and
# each incoming tool is returned at most once even if two canonical
# names happen to be tail-compatible with the same incoming name.
available_by_name: Dict[str, Any] = {}
for tool in available_tools:
client_name, _ = self._extract_tool_info(tool)
if client_name and client_name not in available_by_name:
available_by_name[client_name] = tool
matched: List[Any] = []
used_ids: set = set()
for canonical in tool_names:
tool = available_by_name.get(canonical)
if tool is None:
# Prefer the shortest qualifying name. When several
# incoming tools suffix-match the same canonical (e.g.
# "my_search" and "my_tag_search" both end in "search"),
# the one closest in length to the canonical is the
# least-wrapped and most likely the intended target.
best_name: Optional[str] = None
for client_name in available_by_name:
if not self._name_matches_canonical(client_name, canonical):
continue
if best_name is None or len(client_name) < len(best_name):
best_name = client_name
if best_name is not None:
tool = available_by_name[best_name]
if tool is not None and id(tool) not in used_ids:
matched.append(tool)
used_ids.add(id(tool))
return matched
def extract_user_query(self, messages: List[Dict[str, Any]]) -> str:
"""

View file

@ -7289,6 +7289,7 @@ async def chat_completion( # noqa: PLR0915
and user_api_key_dict.agent_id is not None
):
data["metadata"]["agent_id"] = user_api_key_dict.agent_id
base_llm_response_processor = ProxyBaseLLMRequestProcessing(data=data)
try:
result = await base_llm_response_processor.base_process_llm_request(

View file

@ -139,7 +139,9 @@ class ProviderSpecificModelInfo(TypedDict, total=False):
supports_reasoning: Optional[bool]
supports_url_context: Optional[bool]
supports_none_reasoning_effort: Optional[bool]
supports_minimal_reasoning_effort: Optional[bool]
supports_xhigh_reasoning_effort: Optional[bool]
supports_max_reasoning_effort: Optional[bool]
class SearchContextCostPerQuery(TypedDict, total=False):

View file

@ -5893,9 +5893,15 @@ def _get_model_info_helper( # noqa: PLR0915
supports_none_reasoning_effort=_model_info.get(
"supports_none_reasoning_effort", None
),
supports_minimal_reasoning_effort=_model_info.get(
"supports_minimal_reasoning_effort", None
),
supports_xhigh_reasoning_effort=_model_info.get(
"supports_xhigh_reasoning_effort", None
),
supports_max_reasoning_effort=_model_info.get(
"supports_max_reasoning_effort", None
),
supports_computer_use=_model_info.get("supports_computer_use", None),
search_context_cost_per_query=_model_info.get(
"search_context_cost_per_query", None
@ -8946,6 +8952,12 @@ class ProviderConfigManager:
)
return get_openrouter_image_generation_config(model)
elif LlmProviders.DASHSCOPE == provider:
from litellm.llms.dashscope.image_generation import (
get_dashscope_image_generation_config,
)
return get_dashscope_image_generation_config(model)
return None
@staticmethod

View file

@ -1006,7 +1006,8 @@
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true,
"supports_max_reasoning_effort": true
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"global.anthropic.claude-opus-4-6-v1": {
"cache_creation_input_token_cost": 6.25e-06,
@ -1034,7 +1035,8 @@
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true,
"supports_max_reasoning_effort": true
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"us.anthropic.claude-opus-4-6-v1": {
"cache_creation_input_token_cost": 6.875e-06,
@ -1062,7 +1064,8 @@
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true,
"supports_max_reasoning_effort": true
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"eu.anthropic.claude-opus-4-6-v1": {
"cache_creation_input_token_cost": 6.875e-06,
@ -1090,7 +1093,8 @@
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true,
"supports_max_reasoning_effort": true
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"au.anthropic.claude-opus-4-6-v1": {
"cache_creation_input_token_cost": 6.875e-06,
@ -1118,7 +1122,8 @@
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true,
"supports_max_reasoning_effort": true
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"anthropic.claude-opus-4-7": {
"cache_creation_input_token_cost": 6.25e-06,
@ -1146,7 +1151,9 @@
"supports_vision": true,
"supports_xhigh_reasoning_effort": true,
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
"supports_native_structured_output": true,
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"anthropic.claude-mythos-preview": {
"input_cost_per_token": 0,
@ -1188,7 +1195,9 @@
"supports_vision": true,
"supports_xhigh_reasoning_effort": true,
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
"supports_native_structured_output": true,
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"us.anthropic.claude-opus-4-7": {
"cache_creation_input_token_cost": 6.875e-06,
@ -1216,7 +1225,9 @@
"supports_vision": true,
"supports_xhigh_reasoning_effort": true,
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
"supports_native_structured_output": true,
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"eu.anthropic.claude-opus-4-7": {
"cache_creation_input_token_cost": 6.875e-06,
@ -1244,7 +1255,9 @@
"supports_vision": true,
"supports_xhigh_reasoning_effort": true,
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
"supports_native_structured_output": true,
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"au.anthropic.claude-opus-4-7": {
"cache_creation_input_token_cost": 6.875e-06,
@ -1272,7 +1285,9 @@
"supports_vision": true,
"supports_xhigh_reasoning_effort": true,
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
"supports_native_structured_output": true,
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"anthropic.claude-sonnet-4-6": {
"cache_creation_input_token_cost": 3.75e-06,
@ -1299,7 +1314,8 @@
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
"supports_native_structured_output": true,
"supports_minimal_reasoning_effort": true
},
"global.anthropic.claude-sonnet-4-6": {
"cache_creation_input_token_cost": 3.75e-06,
@ -1326,7 +1342,8 @@
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
"supports_native_structured_output": true,
"supports_minimal_reasoning_effort": true
},
"us.anthropic.claude-sonnet-4-6": {
"cache_creation_input_token_cost": 4.125e-06,
@ -1353,7 +1370,8 @@
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
"supports_native_structured_output": true,
"supports_minimal_reasoning_effort": true
},
"eu.anthropic.claude-sonnet-4-6": {
"cache_creation_input_token_cost": 4.125e-06,
@ -1380,7 +1398,8 @@
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
"supports_native_structured_output": true,
"supports_minimal_reasoning_effort": true
},
"au.anthropic.claude-sonnet-4-6": {
"cache_creation_input_token_cost": 4.125e-06,
@ -1407,7 +1426,8 @@
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
"supports_native_structured_output": true,
"supports_minimal_reasoning_effort": true
},
"anthropic.claude-sonnet-4-20250514-v1:0": {
"cache_creation_input_token_cost": 3.75e-06,
@ -1925,7 +1945,8 @@
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 159,
"supports_max_reasoning_effort": true
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"azure_ai/claude-opus-4-7": {
"input_cost_per_token": 5e-06,
@ -1953,7 +1974,9 @@
"supports_tool_choice": true,
"supports_vision": true,
"supports_xhigh_reasoning_effort": true,
"tool_use_system_prompt_tokens": 159
"tool_use_system_prompt_tokens": 159,
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"azure_ai/claude-opus-4-1": {
"cache_creation_input_token_cost": 1.875e-05,
@ -2017,7 +2040,8 @@
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
"tool_use_system_prompt_tokens": 346,
"supports_minimal_reasoning_effort": true
},
"azure/computer-use-preview": {
"input_cost_per_token": 3e-06,
@ -8923,7 +8947,8 @@
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
"tool_use_system_prompt_tokens": 346,
"supports_minimal_reasoning_effort": true
},
"claude-sonnet-4-5-20250929-v1:0": {
"cache_creation_input_token_cost": 3.75e-06,
@ -9117,7 +9142,8 @@
"us": 1.1,
"fast": 6.0
},
"supports_max_reasoning_effort": true
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"claude-opus-4-6-20260205": {
"cache_creation_input_token_cost": 6.25e-06,
@ -9149,7 +9175,8 @@
"us": 1.1,
"fast": 6.0
},
"supports_max_reasoning_effort": true
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"claude-opus-4-7": {
"cache_creation_input_token_cost": 6.25e-06,
@ -9181,7 +9208,9 @@
"provider_specific_entry": {
"us": 1.1,
"fast": 6.0
}
},
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"claude-opus-4-7-20260416": {
"cache_creation_input_token_cost": 6.25e-06,
@ -9213,7 +9242,9 @@
"provider_specific_entry": {
"us": 1.1,
"fast": 6.0
}
},
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"claude-sonnet-4-20250514": {
"deprecation_date": "2026-05-14",
@ -10366,6 +10397,22 @@
"supports_reasoning": true,
"supports_tool_choice": true
},
"dashscope/qwen-image-2.0": {
"litellm_provider": "dashscope",
"mode": "image_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/models",
"supported_endpoints": [
"/v1/images/generations"
]
},
"dashscope/qwen-image-2.0-pro": {
"litellm_provider": "dashscope",
"mode": "image_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/models",
"supported_endpoints": [
"/v1/images/generations"
]
},
"databricks/databricks-bge-large-en": {
"input_cost_per_token": 1.0003e-07,
"input_dbu_cost_per_token": 1.429e-06,
@ -25082,7 +25129,8 @@
"supports_reasoning": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 159
"tool_use_system_prompt_tokens": 159,
"supports_minimal_reasoning_effort": true
},
"openrouter/anthropic/claude-opus-4.5": {
"cache_creation_input_token_cost": 6.25e-06,
@ -25120,7 +25168,8 @@
"supports_reasoning": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
"tool_use_system_prompt_tokens": 346,
"supports_minimal_reasoning_effort": true
},
"openrouter/anthropic/claude-sonnet-4.5": {
"input_cost_per_image": 0.0048,
@ -30170,7 +30219,8 @@
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true
"supports_vision": true,
"supports_minimal_reasoning_effort": true
},
"vercel_ai_gateway/anthropic/claude-sonnet-4": {
"cache_creation_input_token_cost": 3.75e-06,
@ -31397,7 +31447,8 @@
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
"supports_max_reasoning_effort": true
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"vertex_ai/claude-opus-4-6@default": {
"cache_creation_input_token_cost": 6.25e-06,
@ -31424,7 +31475,8 @@
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
"supports_max_reasoning_effort": true
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"vertex_ai/claude-opus-4-7": {
"cache_creation_input_token_cost": 6.25e-06,
@ -31451,7 +31503,9 @@
"supports_tool_choice": true,
"supports_vision": true,
"supports_xhigh_reasoning_effort": true,
"tool_use_system_prompt_tokens": 346
"tool_use_system_prompt_tokens": 346,
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"vertex_ai/claude-opus-4-7@default": {
"cache_creation_input_token_cost": 6.25e-06,
@ -31478,7 +31532,9 @@
"supports_tool_choice": true,
"supports_vision": true,
"supports_xhigh_reasoning_effort": true,
"tool_use_system_prompt_tokens": 346
"tool_use_system_prompt_tokens": 346,
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": true
},
"vertex_ai/claude-sonnet-4-5": {
"cache_creation_input_token_cost": 3.75e-06,
@ -31530,7 +31586,8 @@
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
"search_context_size_medium": 0.01
}
},
"supports_minimal_reasoning_effort": true
},
"vertex_ai/claude-sonnet-4-5@20250929": {
"cache_creation_input_token_cost": 3.75e-06,
@ -38424,7 +38481,8 @@
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
"search_context_size_medium": 0.01
}
},
"supports_minimal_reasoning_effort": true
},
"duckduckgo/search": {
"litellm_provider": "duckduckgo",

View file

@ -11,6 +11,7 @@ sys.path.insert(
from litellm.litellm_core_utils.prompt_templates.common_utils import (
add_system_prompt_to_messages,
get_file_ids_from_messages,
get_format_from_file_id,
handle_any_messages_to_chat_completion_str_messages_conversion,
split_concatenated_json_objects,
@ -254,3 +255,115 @@ def test_split_concatenated_json_invalid_raises():
"""Completely invalid JSON raises JSONDecodeError."""
with pytest.raises(json.JSONDecodeError):
split_concatenated_json_objects("not json at all")
# ---------------------------------------------------------------------------
# Regression tests for non-OpenAI file content blocks.
#
# `type: "file"` is a public content-block discriminator. Several producers
# (LangChain v1, provider-native shapes, custom user code) emit blocks with
# `type: "file"` but without the OpenAI Chat Completions `file` sub-dict.
# The discovery helpers below are used unconditionally inside
# `AnthropicConfig.validate_environment`, so any crash there surfaces as a
# `500 InternalServerError` before the request is even dispatched.
# ---------------------------------------------------------------------------
def test_get_file_ids_from_messages_skips_langchain_v1_file_block():
"""A LangChain v1 standardized file block must not crash file-id discovery."""
messages = [
{
"role": "user",
"content": [
{"type": "text", "text": "summarise this PDF"},
# LangChain v1 shape produced by `_normalize_messages`.
# No `file` sub-dict: the discriminator is `type: "file"` but
# the payload lives on `base64`/`mime_type` siblings.
{
"type": "file",
"id": "lc_1",
"base64": "JVBERi0xLjQK",
"mime_type": "application/pdf",
"extras": {"file_format": "application/pdf"},
},
],
}
]
assert get_file_ids_from_messages(messages) == []
def test_get_file_ids_from_messages_still_extracts_from_openai_shape():
"""Well-formed OpenAI file blocks still yield their file_id."""
messages = [
{
"role": "user",
"content": [
{"type": "text", "text": "what is this?"},
{"type": "file", "file": {"file_id": "file-abc"}},
],
}
]
assert get_file_ids_from_messages(messages) == ["file-abc"]
def test_get_file_ids_from_messages_mixed_shapes():
"""Mixed OpenAI and non-OpenAI file blocks: extract from the former,
ignore the latter."""
messages = [
{
"role": "user",
"content": [
{"type": "file", "file": {"file_id": "file-keep"}},
{
"type": "file",
"id": "lc_2",
"base64": "AAA",
"mime_type": "application/pdf",
},
],
}
]
assert get_file_ids_from_messages(messages) == ["file-keep"]
def test_get_file_ids_from_messages_file_field_not_dict():
"""`file` set to a non-dict value (e.g. stringified payload) must not crash."""
messages = [
{
"role": "user",
"content": [
{"type": "file", "file": "unexpectedly-a-string"},
],
}
]
assert get_file_ids_from_messages(messages) == []
def test_update_messages_with_model_file_ids_skips_non_openai_file_blocks():
"""`update_messages_with_model_file_ids` is also called on user content
before provider dispatch. It must tolerate non-OpenAI file blocks the same
way."""
langchain_v1_block = {
"type": "file",
"id": "lc_3",
"base64": "AAA",
"mime_type": "application/pdf",
}
messages = [
{
"role": "user",
"content": [
{"type": "text", "text": "hello"},
langchain_v1_block,
],
}
]
updated = update_messages_with_model_file_ids(messages, "model-1", {})
# Messages pass through unchanged when there is no `file` sub-dict to remap.
assert updated == messages

View file

@ -2162,16 +2162,19 @@ class TestTranslateAnthropicOutputFormatToOpenAI:
assert sorted(schema["required"]) == ["age", "email", "name"]
def test_invalid_output_format_returns_none(self):
assert (
self.adapter.translate_anthropic_output_format_to_openai("invalid") is None
)
assert (
self.adapter.translate_anthropic_output_format_to_openai({"type": "text"})
is None
)
assert (
self.adapter.translate_anthropic_output_format_to_openai(
{"type": "json_schema"}
)
is None
)
assert self.adapter.translate_anthropic_output_format_to_openai("invalid") is None
assert self.adapter.translate_anthropic_output_format_to_openai({"type": "text"}) is None
assert self.adapter.translate_anthropic_output_format_to_openai({"type": "json_schema"}) is None
def test_translate_anthropic_tool_choice_none():
"""
Regression test for issue #24443.
tool_choice={"type": "none"} should be translated to "none" for OpenAI format,
not raise a ValueError.
"""
adapter = LiteLLMAnthropicMessagesAdapter()
result = adapter.translate_anthropic_tool_choice_to_openai({"type": "none"})
assert result == "none"

View file

@ -0,0 +1,173 @@
"""
Tests for reasoning_auto_summary support on the native /v1/messages handler.
When reasoning_auto_summary is enabled (via litellm.reasoning_auto_summary or
LITELLM_REASONING_AUTO_SUMMARY env var), the handler injects
thinking.display = "summarized" into the request params for active thinking
modes (type="enabled" or type="adaptive").
"""
import os
import sys
import pytest
from unittest.mock import MagicMock, patch
sys.path.insert(0, os.path.abspath("../../../../.."))
import litellm
from litellm.llms.anthropic.experimental_pass_through.messages.handler import (
anthropic_messages_handler,
)
def _call_handler_and_capture_optional_params(thinking=None, **extra_kwargs):
"""
Call anthropic_messages_handler with an Anthropic model and capture the
anthropic_messages_optional_request_params dict passed to
base_llm_http_handler.anthropic_messages_handler.
Returns the captured dict.
"""
captured = {}
with patch(
"litellm.llms.anthropic.experimental_pass_through.messages.handler."
"base_llm_http_handler"
) as mock_handler, patch(
"litellm.llms.anthropic.experimental_pass_through.messages.handler."
"ProviderConfigManager"
) as mock_pcm:
# Make get_provider_anthropic_messages_config return a non-None config
# so the handler takes the native Anthropic path
mock_pcm.get_provider_anthropic_messages_config.return_value = MagicMock()
mock_handler.anthropic_messages_handler.return_value = MagicMock()
kwargs = dict(extra_kwargs)
if thinking is not None:
kwargs["thinking"] = thinking
try:
anthropic_messages_handler(
max_tokens=1024,
messages=[{"role": "user", "content": "Hello"}],
model="claude-sonnet-4-20250514",
custom_llm_provider="anthropic",
api_key="test-key",
**kwargs,
)
except (ValueError, TypeError, AttributeError):
pass
if mock_handler.anthropic_messages_handler.called:
captured = mock_handler.anthropic_messages_handler.call_args.kwargs.get(
"anthropic_messages_optional_request_params", {}
)
return captured
class TestReasoningAutoSummaryMessages:
"""Tests for thinking.display injection on native /v1/messages handler."""
def test_adaptive_thinking_gets_display_summarized(self):
"""reasoning_auto_summary=True + thinking.type='adaptive' -> display='summarized'."""
with patch.object(litellm, "reasoning_auto_summary", True):
params = _call_handler_and_capture_optional_params(
thinking={"type": "adaptive", "budget_tokens": 5000}
)
thinking = params.get("thinking", {})
assert thinking.get("display") == "summarized"
assert thinking.get("type") == "adaptive"
assert thinking.get("budget_tokens") == 5000
def test_enabled_thinking_gets_display_summarized(self):
"""reasoning_auto_summary=True + thinking.type='enabled' -> display='summarized'."""
with patch.object(litellm, "reasoning_auto_summary", True):
params = _call_handler_and_capture_optional_params(
thinking={"type": "enabled", "budget_tokens": 10000}
)
thinking = params.get("thinking", {})
assert thinking.get("display") == "summarized"
assert thinking.get("type") == "enabled"
def test_disabled_thinking_no_display(self):
"""reasoning_auto_summary=True + thinking.type='disabled' -> display NOT set."""
with patch.object(litellm, "reasoning_auto_summary", True):
params = _call_handler_and_capture_optional_params(
thinking={"type": "disabled"}
)
thinking = params.get("thinking", {})
assert "display" not in thinking
def test_no_injection_when_flag_false(self):
"""reasoning_auto_summary=False + active thinking -> display NOT set."""
with patch.object(litellm, "reasoning_auto_summary", False):
params = _call_handler_and_capture_optional_params(
thinking={"type": "enabled", "budget_tokens": 10000}
)
thinking = params.get("thinking", {})
assert "display" not in thinking
def test_no_thinking_param_no_crash(self):
"""reasoning_auto_summary=True but no thinking param -> nothing changes."""
with patch.object(litellm, "reasoning_auto_summary", True):
params = _call_handler_and_capture_optional_params()
thinking = params.get("thinking")
if thinking is not None:
assert "display" not in thinking
def test_env_var_enables_auto_summary(self):
"""LITELLM_REASONING_AUTO_SUMMARY=true env var enables the feature."""
with patch.object(litellm, "reasoning_auto_summary", False), patch.dict(
os.environ, {"LITELLM_REASONING_AUTO_SUMMARY": "true"}
):
params = _call_handler_and_capture_optional_params(
thinking={"type": "adaptive", "budget_tokens": 5000}
)
thinking = params.get("thinking", {})
assert thinking.get("display") == "summarized"
def test_existing_display_summarized_preserved(self):
"""User already passes display='summarized' -> preserved as-is."""
with patch.object(litellm, "reasoning_auto_summary", True):
params = _call_handler_and_capture_optional_params(
thinking={
"type": "enabled",
"budget_tokens": 10000,
"display": "summarized",
}
)
thinking = params.get("thinking", {})
assert thinking.get("display") == "summarized"
def test_existing_display_summarized_without_flag(self):
"""User passes display='summarized' + flag=False -> preserved as-is."""
with patch.object(litellm, "reasoning_auto_summary", False):
params = _call_handler_and_capture_optional_params(
thinking={
"type": "enabled",
"budget_tokens": 10000,
"display": "summarized",
}
)
thinking = params.get("thinking", {})
assert thinking.get("display") == "summarized"
def test_omitted_overridden_to_summarized(self):
"""User passes display='omitted' + reasoning_auto_summary=True -> overridden.
Documents current behavior: the code unconditionally sets
display='summarized' when auto_summary is enabled and thinking is active,
regardless of any pre-existing display value.
"""
with patch.object(litellm, "reasoning_auto_summary", True):
params = _call_handler_and_capture_optional_params(
thinking={
"type": "enabled",
"budget_tokens": 10000,
"display": "omitted",
}
)
thinking = params.get("thinking", {})
assert thinking.get("display") == "summarized"

View file

@ -0,0 +1,287 @@
"""
Tests for reasoning effort capability fields and normalize_reasoning_effort_value.
Covers:
- Commit 1: get_model_info returns supports_minimal/supports_max fields
- Commit 2: Model registry entries have correct reasoning effort fields
- Commit 3: normalize_reasoning_effort_value degradation chains + adapter translation
"""
import json
import os
from typing import Any, Dict, Optional
from unittest.mock import patch
import pytest
from litellm.llms.anthropic.experimental_pass_through.utils import (
normalize_reasoning_effort_value,
)
from litellm.utils import get_model_info
def _load_model_registry() -> Dict[str, Any]:
"""Load the root model_prices_and_context_window.json."""
json_path = os.path.join(
os.path.dirname(__file__),
"../../../../../model_prices_and_context_window.json",
)
with open(json_path) as f:
return json.load(f)
# ---------------------------------------------------------------------------
# Commit 1: get_model_info returns supports_minimal and supports_max fields
# ---------------------------------------------------------------------------
class TestGetModelInfoReasoningEffortFields:
"""get_model_info should expose supports_minimal_reasoning_effort and
supports_max_reasoning_effort from the model registry."""
def test_opus_4_6_has_supports_minimal(self):
info = get_model_info("claude-opus-4-6")
assert "supports_minimal_reasoning_effort" in info
def test_opus_4_6_has_supports_max(self):
info = get_model_info("claude-opus-4-6")
assert "supports_max_reasoning_effort" in info
def test_opus_4_7_has_supports_minimal(self):
info = get_model_info("claude-opus-4-7")
assert "supports_minimal_reasoning_effort" in info
def test_opus_4_7_has_supports_max(self):
info = get_model_info("claude-opus-4-7")
assert "supports_max_reasoning_effort" in info
# ---------------------------------------------------------------------------
# Commit 2: JSON registry has correct reasoning effort fields
# ---------------------------------------------------------------------------
class TestModelRegistryReasoningEffortFields:
"""Verify specific models have the expected reasoning effort capability
values in the JSON registry file."""
@pytest.fixture(autouse=True)
def _load_registry(self):
self.registry = _load_model_registry()
def test_opus_4_7_supports_max(self):
entry = self.registry["claude-opus-4-7"]
assert entry.get("supports_max_reasoning_effort") is True
def test_opus_4_6_supports_max(self):
entry = self.registry["claude-opus-4-6"]
assert entry.get("supports_max_reasoning_effort") is True
def test_opus_4_7_supports_minimal(self):
entry = self.registry["claude-opus-4-7"]
assert entry.get("supports_minimal_reasoning_effort") is True
def test_opus_4_6_supports_minimal(self):
entry = self.registry["claude-opus-4-6"]
assert entry.get("supports_minimal_reasoning_effort") is True
def test_sonnet_4_6_supports_minimal(self):
entry = self.registry["anthropic.claude-sonnet-4-6"]
assert entry.get("supports_minimal_reasoning_effort") is True
def test_bedrock_opus_4_7_supports_max(self):
entry = self.registry["anthropic.claude-opus-4-7"]
assert entry.get("supports_max_reasoning_effort") is True
assert entry.get("supports_minimal_reasoning_effort") is True
def test_vertex_opus_4_7_supports_max(self):
entry = self.registry["vertex_ai/claude-opus-4-7"]
assert entry.get("supports_max_reasoning_effort") is True
assert entry.get("supports_minimal_reasoning_effort") is True
def test_vertex_opus_4_6_supports_max(self):
entry = self.registry["vertex_ai/claude-opus-4-6"]
assert entry.get("supports_max_reasoning_effort") is True
assert entry.get("supports_minimal_reasoning_effort") is True
def test_azure_ai_opus_4_6_supports_minimal(self):
entry = self.registry["azure_ai/claude-opus-4-6"]
assert entry.get("supports_minimal_reasoning_effort") is True
def test_azure_ai_opus_4_7_supports_max(self):
entry = self.registry["azure_ai/claude-opus-4-7"]
assert entry.get("supports_max_reasoning_effort") is True
assert entry.get("supports_minimal_reasoning_effort") is True
# ---------------------------------------------------------------------------
# Commit 3: normalize_reasoning_effort_value
# ---------------------------------------------------------------------------
def _mock_model_info(**flags):
"""Return a mock model_info dict with given capability flags."""
return flags
class TestNormalizeReasoningEffortValue:
"""Test degradation chains for normalize_reasoning_effort_value."""
# --- "max" degradation chain ---
def test_max_stays_max_when_supported(self):
with patch(
"litellm.utils.get_model_info",
return_value=_mock_model_info(
supports_max_reasoning_effort=True,
supports_xhigh_reasoning_effort=True,
),
):
assert normalize_reasoning_effort_value("max", model="test") == "max"
def test_max_degrades_to_xhigh(self):
with patch(
"litellm.utils.get_model_info",
return_value=_mock_model_info(
supports_max_reasoning_effort=False,
supports_xhigh_reasoning_effort=True,
),
):
assert normalize_reasoning_effort_value("max", model="test") == "xhigh"
def test_max_degrades_to_high(self):
with patch(
"litellm.utils.get_model_info",
return_value=_mock_model_info(
supports_max_reasoning_effort=False,
supports_xhigh_reasoning_effort=False,
),
):
assert normalize_reasoning_effort_value("max", model="test") == "high"
# --- "xhigh" degradation chain ---
def test_xhigh_stays_xhigh_when_supported(self):
with patch(
"litellm.utils.get_model_info",
return_value=_mock_model_info(supports_xhigh_reasoning_effort=True),
):
assert normalize_reasoning_effort_value("xhigh", model="test") == "xhigh"
def test_xhigh_degrades_to_high(self):
with patch(
"litellm.utils.get_model_info",
return_value=_mock_model_info(supports_xhigh_reasoning_effort=False),
):
assert normalize_reasoning_effort_value("xhigh", model="test") == "high"
# --- "minimal" degradation chain ---
def test_minimal_stays_minimal_when_supported(self):
with patch(
"litellm.utils.get_model_info",
return_value=_mock_model_info(supports_minimal_reasoning_effort=True),
):
assert (
normalize_reasoning_effort_value("minimal", model="test") == "minimal"
)
def test_minimal_degrades_to_low(self):
with patch(
"litellm.utils.get_model_info",
return_value=_mock_model_info(supports_minimal_reasoning_effort=False),
):
assert normalize_reasoning_effort_value("minimal", model="test") == "low"
# --- passthrough values ---
def test_high_passes_through(self):
assert normalize_reasoning_effort_value("high", model="test") == "high"
def test_medium_passes_through(self):
assert normalize_reasoning_effort_value("medium", model="test") == "medium"
def test_low_passes_through(self):
assert normalize_reasoning_effort_value("low", model="test") == "low"
# --- exception fallback ---
def test_exception_fallback_uses_empty_model_info(self):
"""When get_model_info raises, treat model_info as {} (no capabilities)."""
with patch(
"litellm.utils.get_model_info",
side_effect=Exception("model not found"),
):
# "max" with no capabilities -> "high"
assert normalize_reasoning_effort_value("max", model="unknown") == "high"
# "minimal" with no capabilities -> "low"
assert normalize_reasoning_effort_value("minimal", model="unknown") == "low"
# ---------------------------------------------------------------------------
# Commit 3: Adapter translation — adaptive thinking + output_config.effort
# ---------------------------------------------------------------------------
class TestAdapterAdaptiveThinking:
"""Test that adaptive thinking type maps correctly through the adapters."""
def test_messages_adapter_adaptive_returns_medium_default(self):
"""Adaptive thinking returns 'medium' as default reasoning_effort."""
from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import (
LiteLLMAnthropicMessagesAdapter,
)
adapter = LiteLLMAnthropicMessagesAdapter()
result = adapter.translate_anthropic_thinking_to_reasoning_effort(
{"type": "adaptive"}
)
assert result == "medium"
def test_messages_adapter_adaptive_overridden_by_output_config(self):
"""For adaptive thinking, output_config.effort overrides reasoning_effort."""
from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import (
LiteLLMAnthropicMessagesAdapter,
)
from litellm.types.llms.anthropic import AnthropicMessagesRequest
adapter = LiteLLMAnthropicMessagesAdapter()
request = AnthropicMessagesRequest(
model="test-model",
messages=[{"role": "user", "content": "hello"}],
max_tokens=1024,
thinking={"type": "adaptive"},
output_config={"effort": "high"},
)
openai_kwargs, _ = adapter.translate_anthropic_to_openai(request)
# reasoning_effort should be set (either as string or dict with effort)
re = openai_kwargs.get("reasoning_effort")
if isinstance(re, dict):
assert re["effort"] == "high"
else:
assert re == "high"
def test_responses_adapter_adaptive_with_output_config(self):
"""Responses adapter: adaptive thinking + output_config.effort."""
from litellm.llms.anthropic.experimental_pass_through.responses_adapters.transformation import (
LiteLLMAnthropicToResponsesAPIAdapter,
)
result = LiteLLMAnthropicToResponsesAPIAdapter.translate_thinking_to_reasoning(
thinking={"type": "adaptive"},
output_config={"effort": "xhigh"},
)
assert result is not None
assert result["effort"] == "xhigh"
def test_responses_adapter_adaptive_default_medium(self):
"""Responses adapter: adaptive thinking without output_config defaults to medium."""
from litellm.llms.anthropic.experimental_pass_through.responses_adapters.transformation import (
LiteLLMAnthropicToResponsesAPIAdapter,
)
result = LiteLLMAnthropicToResponsesAPIAdapter.translate_thinking_to_reasoning(
thinking={"type": "adaptive"},
)
assert result is not None
assert result["effort"] == "medium"

View file

@ -0,0 +1,151 @@
"""
Regression tests for is_model_gpt_5_model() in both OpenAI and Azure GPT-5 config
classes.
Background
----------
In v1.82.3 a substring check was introduced::
return "gpt-5" in model and "gpt-5-chat" not in model
This inadvertently treated versioned chat models like ``gpt-5.3-chat`` and
``gpt-5.1-chat`` as *non*-GPT-5 models, because the string ``"gpt-5-chat"`` is
a substring of ``"gpt-5.3-chat"``. Those models were then routed through the
regular Azure chat path which does not suppress ``parallel_tool_calls``, causing
Azure to return ``finish_reason="stop"`` together with tool_calls and breaking
n8n AI-agent workflows.
There are two distinct families:
* **gpt-5-chat family** (``gpt-5-chat``, ``gpt-5-chat-latest``,
``gpt-5-chat-2025-08-07``, ) regular chat models that support ``temperature``
and ``tool_choice`` but NOT ``reasoning_effort``. Must NOT be on the GPT-5
reasoning path.
* **Versioned chat models** (``gpt-5.1-chat``, ``gpt-5.2-chat``,
``gpt-5.3-chat``, ) ARE GPT-5 reasoning models and must stay on the GPT-5
path.
The fix uses a prefix check (``startswith("gpt-5-chat")``) on the normalised model
name instead of a substring check, which correctly distinguishes the two families.
"""
import pytest
from litellm.llms.openai.chat.gpt_5_transformation import OpenAIGPT5Config
from litellm.llms.azure.chat.gpt_5_transformation import AzureOpenAIGPT5Config
# ---------------------------------------------------------------------------
# Parametrized fixtures
# ---------------------------------------------------------------------------
# Models that MUST be classified as GPT-5 (routed through GPT-5 reasoning path)
GPT5_MODELS = [
"gpt-5",
"gpt-5.1",
"gpt-5.2",
"gpt-5.3",
"gpt-5.4",
"gpt-5.1-chat", # versioned chat — THE KEY REGRESSION CASE
"gpt-5.2-chat", # versioned chat — also a regression case
"gpt-5.3-chat", # versioned chat — THE KEY REGRESSION CASE
"gpt-5.2-chat-latest", # versioned chat with date suffix
"gpt-5.1-codex",
"gpt-5.1-codex-mini",
"gpt-5.1-mini",
"gpt-5-nano",
"gpt-5-mini",
"gpt-5-codex",
]
# Models that must NOT be classified as GPT-5 (regular chat path)
NON_GPT5_MODELS = [
"gpt-5-chat", # gpt-5-chat family — regular chat path
"gpt-5-chat-latest", # gpt-5-chat family with alias suffix
"gpt-5-chat-2025-08-07", # gpt-5-chat family with date suffix
"gpt-4",
"gpt-4o",
"gpt-4-turbo",
"gpt-3.5-turbo",
"o1",
"o3",
"o3-mini",
]
# ---------------------------------------------------------------------------
# OpenAIGPT5Config
# ---------------------------------------------------------------------------
class TestOpenAIGPT5ConfigIsModelGpt5Model:
@pytest.mark.parametrize("model", GPT5_MODELS)
def test_gpt5_models_are_classified_as_gpt5(self, model: str):
assert OpenAIGPT5Config.is_model_gpt_5_model(
model
), f"Expected '{model}' to be classified as a GPT-5 model"
@pytest.mark.parametrize("model", NON_GPT5_MODELS)
def test_non_gpt5_models_are_not_classified_as_gpt5(self, model: str):
assert not OpenAIGPT5Config.is_model_gpt_5_model(
model
), f"Expected '{model}' NOT to be classified as a GPT-5 model"
def test_versioned_chat_models_are_not_excluded_by_prefix(self):
"""Core regression guard: gpt-5-chat prefix must not match versioned models."""
versioned_chat_models = ["gpt-5.1-chat", "gpt-5.2-chat", "gpt-5.3-chat"]
for model in versioned_chat_models:
assert OpenAIGPT5Config.is_model_gpt_5_model(
model
), f"Regression: '{model}' was incorrectly excluded from GPT-5 path"
def test_gpt5_chat_family_is_excluded(self):
"""gpt-5-chat family should stay on the regular chat path."""
for model in ["gpt-5-chat", "gpt-5-chat-latest", "gpt-5-chat-2025-08-07"]:
assert not OpenAIGPT5Config.is_model_gpt_5_model(
model
), f"Expected '{model}' (gpt-5-chat family) NOT to be on the GPT-5 path"
# ---------------------------------------------------------------------------
# AzureOpenAIGPT5Config
# ---------------------------------------------------------------------------
class TestAzureOpenAIGPT5ConfigIsModelGpt5Model:
@pytest.mark.parametrize("model", GPT5_MODELS)
def test_gpt5_models_are_classified_as_gpt5(self, model: str):
assert AzureOpenAIGPT5Config.is_model_gpt_5_model(
model
), f"Expected Azure '{model}' to be classified as a GPT-5 model"
@pytest.mark.parametrize("model", NON_GPT5_MODELS)
def test_non_gpt5_models_are_not_classified_as_gpt5(self, model: str):
assert not AzureOpenAIGPT5Config.is_model_gpt_5_model(
model
), f"Expected Azure '{model}' NOT to be classified as a GPT-5 model"
def test_versioned_chat_models_are_not_excluded_by_prefix(self):
"""Core regression guard: gpt-5-chat prefix must not match versioned models."""
versioned_chat_models = ["gpt-5.1-chat", "gpt-5.2-chat", "gpt-5.3-chat"]
for model in versioned_chat_models:
assert AzureOpenAIGPT5Config.is_model_gpt_5_model(
model
), f"Regression: Azure '{model}' was incorrectly excluded from GPT-5 path"
def test_gpt5_chat_family_is_excluded(self):
"""gpt-5-chat family should stay on the regular chat path."""
for model in ["gpt-5-chat", "gpt-5-chat-latest", "gpt-5-chat-2025-08-07"]:
assert not AzureOpenAIGPT5Config.is_model_gpt_5_model(
model
), f"Expected Azure '{model}' (gpt-5-chat family) NOT to be on the GPT-5 path"
def test_gpt5_series_routing_prefix_is_always_classified_as_gpt5(self):
"""Models using the gpt5_series/ manual-routing prefix must always match."""
series_models = ["gpt5_series/my-deployment", "gpt5_series/prod"]
for model in series_models:
assert AzureOpenAIGPT5Config.is_model_gpt_5_model(
model
), f"Azure '{model}' with gpt5_series/ prefix should be classified as GPT-5"

View file

@ -8,6 +8,7 @@ import sys
import pytest
from litellm.llms.ovhcloud.utils import OVHCloudException
from litellm.utils import get_optional_params
sys.path.insert(
0, os.path.abspath("../../../../..")
@ -144,6 +145,38 @@ class TestOVHCloudConfig:
assert error.message == "Test error"
assert error.status_code == 400
@pytest.mark.parametrize(
"model",
[
"Meta-Llama-3_3-70B-Instruct",
"Meta-Llama-3_1-70B-Instruct",
"Mixtral-8x7B-Instruct-v0.1",
"gpt-oss-120b",
"some-model-not-in-the-cost-map",
],
)
def test_tools_not_filtered_by_static_model_map(self, model):
"""
OVHCloud AI Endpoints are OpenAI-compatible; tools/tool_choice must pass
through for any model. The server is responsible for rejecting unsupported
tool calls LiteLLM must not strip them based on a stale static catalog.
"""
params = get_optional_params(
model=model,
custom_llm_provider="ovhcloud",
tools=[
{
"type": "function",
"function": {"name": "x", "parameters": {}},
}
],
tool_choice="auto",
)
assert "tools" in params
assert "tool_choice" in params
def test_ovhcloud_integration():
import os

View file

@ -967,6 +967,94 @@ class TestMediaResolution:
assert "mediaResolution" not in result["generationConfig"]
# Tests for VideoMetadata support across all Gemini models (Issue #25474)
class TestVideoMetadataAllGeminiModels:
"""Tests that video_metadata (fps, start_offset, end_offset) works for all Gemini models"""
def _make_video_messages(self, video_metadata: dict) -> list:
return [
{
"role": "user",
"content": [
{"type": "text", "text": "Analyze this video"},
{
"type": "file",
"file": {
"file_id": "gs://bucket/video.mp4",
"format": "video/mp4",
"video_metadata": video_metadata,
},
},
],
}
]
def _get_file_part(self, contents: list) -> dict:
for part in contents[0]["parts"]:
if "file_data" in part:
return part
raise AssertionError("No file part found in contents")
def test_video_metadata_fps_gemini_2_5_flash(self):
"""Gemini 2.5 Flash: fps in video_metadata should be forwarded (Issue #25474)"""
messages = self._make_video_messages({"fps": 5})
contents = _gemini_convert_messages_with_history(
messages=messages, model="gemini-2.5-flash"
)
file_part = self._get_file_part(contents)
assert "video_metadata" in file_part
assert file_part["video_metadata"]["fps"] == 5
def test_video_metadata_fps_gemini_2_5_pro(self):
"""Gemini 2.5 Pro: fps in video_metadata should be forwarded (Issue #25474)"""
messages = self._make_video_messages({"fps": 10})
contents = _gemini_convert_messages_with_history(
messages=messages, model="gemini-2.5-pro"
)
file_part = self._get_file_part(contents)
assert "video_metadata" in file_part
assert file_part["video_metadata"]["fps"] == 10
def test_video_metadata_offsets_gemini_2_5_flash(self):
"""Gemini 2.5 Flash: start_offset/end_offset converted to camelCase (Issue #25474)"""
messages = self._make_video_messages(
{"start_offset": "5s", "end_offset": "30s"}
)
contents = _gemini_convert_messages_with_history(
messages=messages, model="gemini-2.5-flash"
)
file_part = self._get_file_part(contents)
assert "video_metadata" in file_part
vm = file_part["video_metadata"]
assert vm["startOffset"] == "5s"
assert vm["endOffset"] == "30s"
def test_video_metadata_all_fields_gemini_2_5_flash(self):
"""Gemini 2.5 Flash: all video_metadata fields forwarded correctly (Issue #25474)"""
messages = self._make_video_messages(
{"fps": 5, "start_offset": "10s", "end_offset": "60s"}
)
contents = _gemini_convert_messages_with_history(
messages=messages, model="gemini-2.5-flash"
)
file_part = self._get_file_part(contents)
assert "video_metadata" in file_part
vm = file_part["video_metadata"]
assert vm["fps"] == 5
assert vm["startOffset"] == "10s"
assert vm["endOffset"] == "60s"
def test_video_metadata_gemini_1_5_pro(self):
"""Gemini 1.5 Pro: video_metadata should also be forwarded (Issue #25474)"""
messages = self._make_video_messages({"fps": 2})
contents = _gemini_convert_messages_with_history(
messages=messages, model="gemini-1.5-pro"
)
file_part = self._get_file_part(contents)
assert "video_metadata" in file_part
assert file_part["video_metadata"]["fps"] == 2
def test_convert_tool_response_with_base64_image():
"""Test tool response with base64 data URI image."""
# Create a small test image (1x1 red pixel PNG)

View file

@ -3476,8 +3476,8 @@ def test_new_detail_levels():
assert file_part["media_resolution"] == {"level": "MEDIA_RESOLUTION_MEDIUM"}
def test_video_metadata_only_for_gemini_3():
"""Test that video_metadata is only applied for Gemini 3+ models (Issue #19026)"""
def test_video_metadata_supported_for_all_gemini_models():
"""Test that video_metadata is applied for all Gemini models (Issue #25474)"""
from litellm.llms.vertex_ai.gemini.transformation import (
_gemini_convert_messages_with_history,
)
@ -3499,39 +3499,29 @@ def test_video_metadata_only_for_gemini_3():
}
]
# Test with Gemini 1.5 (should not have video_metadata or media_resolution)
contents_1_5 = _gemini_convert_messages_with_history(
messages=messages, model="gemini-1.5-pro"
)
for model in ["gemini-1.5-pro", "gemini-2.5-flash", "gemini-2.5-pro", "gemini-3-pro-preview"]:
contents = _gemini_convert_messages_with_history(messages=messages, model=model)
file_part_1_5 = None
for part in contents_1_5[0]["parts"]:
if "file_data" in part:
file_part_1_5 = part
break
file_part = None
for part in contents[0]["parts"]:
if "file_data" in part:
file_part = part
break
assert file_part_1_5 is not None
assert (
"media_resolution" not in file_part_1_5
), "Gemini 1.5 should not have media_resolution"
assert (
"video_metadata" not in file_part_1_5
), "Gemini 1.5 should not have video_metadata"
assert file_part is not None, f"{model}: file part should exist"
assert "video_metadata" in file_part, f"{model}: video_metadata should be present"
assert file_part["video_metadata"]["fps"] == 5, f"{model}: fps should be 5"
# Test with Gemini 3 (should have both)
contents_3 = _gemini_convert_messages_with_history(
messages=messages, model="gemini-3-pro-preview"
)
# Per-part media_resolution is Gemini 3+ only; 2.x uses generation_config global
for model in ["gemini-3-pro-preview"]:
contents = _gemini_convert_messages_with_history(messages=messages, model=model)
file_part = next(p for p in contents[0]["parts"] if "file_data" in p)
assert "media_resolution" in file_part, f"{model}: media_resolution should be present"
file_part_3 = None
for part in contents_3[0]["parts"]:
if "file_data" in part:
file_part_3 = part
break
assert file_part_3 is not None
assert "media_resolution" in file_part_3, "Gemini 3 should have media_resolution"
assert "video_metadata" in file_part_3, "Gemini 3 should have video_metadata"
for model in ["gemini-1.5-pro", "gemini-2.5-flash", "gemini-2.5-pro"]:
contents = _gemini_convert_messages_with_history(messages=messages, model=model)
file_part = next(p for p in contents[0]["parts"] if "file_data" in p)
assert "media_resolution" not in file_part, f"{model}: per-part media_resolution should not be set"
def test_chunk_parser_handles_prompt_feedback_block():

View file

@ -450,3 +450,193 @@ async def test_semantic_filter_hook_skips_no_tools():
# Should return None (no modification)
assert result is None, "Hook should skip requests without tools"
print("✅ Hook correctly skips requests without tools")
class TestGetToolsByNames:
"""
Regression coverage for SemanticMCPToolFilter._get_tools_by_names
name-matching behavior (issue #26078).
The canonical name stored in the router is what the proxy's MCP
registry emits (e.g. ``fc_web_search-firecrawl_scrape``). Some MCP
clients notably opencode wrap every tool name with their own
additive namespace prefix before sending it back in ``tools[]``, so
the incoming name is ``litellm_fc_web_search-firecrawl_scrape``.
Exact-equality matching against the canonical dropped every such
tool, the proxy forwarded ``tools: []`` with ``tool_choice: auto``,
and strict upstream providers returned 400.
"""
def _make_filter(self):
from litellm.proxy._experimental.mcp_server.semantic_tool_filter import (
SemanticMCPToolFilter,
)
return SemanticMCPToolFilter(
embedding_model="text-embedding-3-small",
litellm_router_instance=Mock(),
top_k=5,
similarity_threshold=0.3,
enabled=True,
)
def test_exact_match_unchanged(self):
"""Incoming name equals canonical — the historical path still works."""
filter_instance = self._make_filter()
available_tools = [
{"name": "get_weather", "description": "fetch weather"},
{"name": "send_email", "description": "send mail"},
]
matched = filter_instance._get_tools_by_names(
["send_email"], available_tools
)
assert len(matched) == 1
assert matched[0]["name"] == "send_email"
def test_client_prefix_with_underscore_separator(self):
"""Client wraps canonical with ``<alias>_`` (opencode pattern)."""
filter_instance = self._make_filter()
canonical = "fc_web_search-firecrawl_scrape"
client_name = "litellm_" + canonical
available_tools = [{"name": client_name, "description": "scrape"}]
matched = filter_instance._get_tools_by_names(
[canonical], available_tools
)
assert len(matched) == 1
# Must return the incoming tool unchanged so the client-facing
# name survives, otherwise tool-call round-trips break client-side.
assert matched[0]["name"] == client_name
def test_client_prefix_with_dash_separator(self):
"""Some clients use dash as alias separator; accept that too."""
filter_instance = self._make_filter()
canonical = "weather_svc-get_weather"
available_tools = [
{"name": "mcp-" + canonical, "description": "weather"}
]
matched = filter_instance._get_tools_by_names(
[canonical], available_tools
)
assert len(matched) == 1
assert matched[0]["name"] == "mcp-" + canonical
def test_suffix_without_separator_does_not_match(self):
"""
A bare-substring suffix must not match ``rain_gear`` is not a
namespaced version of canonical ``ear`` and the user would be
surprised to see it selected.
"""
filter_instance = self._make_filter()
available_tools = [{"name": "rain_gear", "description": "raincoat"}]
matched = filter_instance._get_tools_by_names(["ear"], available_tools)
assert matched == []
def test_exact_match_preferred_over_prefixed(self):
"""
When both a bare canonical and a client-prefixed variant are
present, the bare one wins so ordering is stable.
"""
filter_instance = self._make_filter()
canonical = "search"
available_tools = [
{"name": canonical, "description": "plain"},
{"name": "litellm_" + canonical, "description": "wrapped"},
]
matched = filter_instance._get_tools_by_names(
[canonical], available_tools
)
assert len(matched) == 1
assert matched[0]["name"] == canonical
def test_same_tool_not_returned_twice(self):
"""
Two distinct canonicals that both suffix-match the same incoming
tool must not produce a duplicate in the output list.
``fs-read_file`` and ``api-fs-read_file`` are both valid
separator-anchored suffixes of ``litellm_api-fs-read_file``.
"""
filter_instance = self._make_filter()
available_tools = [
{"name": "litellm_api-fs-read_file", "description": "read"}
]
matched = filter_instance._get_tools_by_names(
["fs-read_file", "api-fs-read_file"], available_tools
)
assert len(matched) == 1
def test_suffix_fallback_prefers_shortest_candidate(self):
"""
When no exact match exists and several incoming tools
suffix-match the same canonical, the one closest in length to
the canonical (i.e. the least-wrapped) should be chosen.
"""
filter_instance = self._make_filter()
canonical = "svc-search"
available_tools = [
{"name": "my_tag_" + canonical, "description": "tag search"},
{"name": "my_" + canonical, "description": "plain search"},
]
matched = filter_instance._get_tools_by_names(
[canonical], available_tools
)
assert len(matched) == 1
assert matched[0]["name"] == "my_" + canonical
def test_ordering_follows_router_output(self):
"""Returned tools follow the order the semantic router chose."""
filter_instance = self._make_filter()
available_tools = [
{"name": "litellm_fs-read", "description": "read"},
{"name": "litellm_fs-write", "description": "write"},
{"name": "litellm_fs-delete", "description": "delete"},
]
matched = filter_instance._get_tools_by_names(
["fs-write", "fs-delete", "fs-read"], available_tools
)
names = [t["name"] for t in matched]
assert names == [
"litellm_fs-write",
"litellm_fs-delete",
"litellm_fs-read",
]
def test_does_not_collide_with_local_function_on_unprefixed_canonical(self):
"""
Guard against the collision @krrish-berri-2 flagged on #26117:
if the canonical name from the router is not server-prefixed
(i.e. does not contain ``MCP_TOOL_PREFIX_SEPARATOR``), suffix
matching must not kick in. Otherwise an unrelated local user
function whose name happens to end in the canonical substring
would be spuriously selected.
"""
filter_instance = self._make_filter()
available_tools = [
{
"name": "my_firecrawl_scrape",
"description": "unrelated local function",
},
]
matched = filter_instance._get_tools_by_names(
["firecrawl_scrape"], # no MCP_TOOL_PREFIX_SEPARATOR in canonical
available_tools,
)
assert matched == []

View file

@ -0,0 +1,401 @@
"""
Unit tests for DashScope image generation support (qwen-image-2.0, qwen-image-2.0-pro).
Run in docker: pytest tests/test_litellm/test_dashscope_image_generation.py -v
"""
import json
from unittest.mock import MagicMock, patch
import httpx
import pytest
import litellm
from litellm.llms.dashscope.image_generation.transformation import (
DashScopeImageGenerationConfig,
DEFAULT_API_BASE,
)
from litellm.types.utils import ImageObject, ImageResponse
from litellm.utils import get_llm_provider
# ---------------------------------------------------------------------------
# 1. Provider detection
# ---------------------------------------------------------------------------
@pytest.mark.parametrize(
"model_string",
[
"dashscope/qwen-image-2.0",
"dashscope/qwen-image-2.0-pro",
],
)
def test_get_llm_provider_returns_dashscope(model_string: str):
model, provider, _, _ = get_llm_provider(model_string)
assert provider == "dashscope", f"Expected 'dashscope', got '{provider}'"
assert "qwen-image" in model
# ---------------------------------------------------------------------------
# 2. Model info: mode == "image_generation"
# ---------------------------------------------------------------------------
@pytest.mark.parametrize(
"model_string, custom_provider",
[
("dashscope/qwen-image-2.0", "dashscope"),
("dashscope/qwen-image-2.0-pro", "dashscope"),
],
)
def test_get_model_info_mode_is_image_generation(
model_string: str, custom_provider: str
):
import os
prev_env = os.environ.get("LITELLM_LOCAL_MODEL_COST_MAP")
prev_model_cost = litellm.model_cost
try:
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
info = litellm.get_model_info(
model=model_string, custom_llm_provider=custom_provider
)
assert (
info["mode"] == "image_generation"
), f"Expected mode='image_generation', got '{info['mode']}'"
finally:
if prev_env is None:
os.environ.pop("LITELLM_LOCAL_MODEL_COST_MAP", None)
else:
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = prev_env
litellm.model_cost = prev_model_cost
# ---------------------------------------------------------------------------
# 3. Request transformation
# ---------------------------------------------------------------------------
class TestDashScopeImageGenerationConfig:
def setup_method(self):
self.cfg = DashScopeImageGenerationConfig()
def test_get_complete_url_default(self):
url = self.cfg.get_complete_url(None, None, "qwen-image-2.0", {}, {})
assert url == DEFAULT_API_BASE
def test_get_complete_url_custom(self):
custom = "https://custom.endpoint/generate"
url = self.cfg.get_complete_url(custom, None, "qwen-image-2.0", {}, {})
assert url == custom
def test_validate_environment_sets_auth_header(self):
headers = self.cfg.validate_environment(
headers={},
model="qwen-image-2.0",
messages=[],
optional_params={},
litellm_params={},
api_key="sk-test-key",
)
assert headers["Authorization"] == "Bearer sk-test-key"
assert headers["Content-Type"] == "application/json"
def test_validate_environment_raises_without_key(self):
with patch(
"litellm.llms.dashscope.image_generation.transformation.get_secret_str",
return_value=None,
):
with pytest.raises(ValueError, match="DASHSCOPE_API_KEY"):
self.cfg.validate_environment(
headers={},
model="qwen-image-2.0",
messages=[],
optional_params={},
litellm_params={},
api_key=None,
)
def test_transform_request_structure(self):
req = self.cfg.transform_image_generation_request(
model="qwen-image-2.0",
prompt="a puppy on green grass",
optional_params={"size": "1024*1024"},
litellm_params={},
headers={},
)
assert req["model"] == "qwen-image-2.0"
messages = req["input"]["messages"]
assert len(messages) == 1
assert messages[0]["role"] == "user"
assert messages[0]["content"][0]["text"] == "a puppy on green grass"
assert req["parameters"]["size"] == "1024*1024"
def test_transform_request_empty_params(self):
req = self.cfg.transform_image_generation_request(
model="qwen-image-2.0-pro",
prompt="sunset over the ocean",
optional_params={},
litellm_params={},
headers={},
)
assert req["parameters"] == {}
# ---------------------------------------------------------------------------
# 4. Response transformation
# ---------------------------------------------------------------------------
def _make_mock_response(self, image_url: str) -> httpx.Response:
body = {
"status_code": 200,
"request_id": "test-request-id",
"output": {
"choices": [
{
"finish_reason": "stop",
"message": {
"role": "assistant",
"content": [{"image": image_url}],
},
}
]
},
"usage": {
"input_tokens": 0,
"output_tokens": 0,
"width": 1024,
"height": 1024,
"image_count": 1,
},
}
mock_resp = MagicMock(spec=httpx.Response)
mock_resp.status_code = 200
mock_resp.headers = {}
mock_resp.json.return_value = body
return mock_resp
def test_transform_response_extracts_url(self):
image_url = "https://example.oss.aliyuncs.com/generated/test.png"
mock_resp = self._make_mock_response(image_url)
model_response = ImageResponse()
result = self.cfg.transform_image_generation_response(
model="qwen-image-2.0",
raw_response=mock_resp,
model_response=model_response,
logging_obj=MagicMock(),
request_data={},
optional_params={},
litellm_params={},
encoding=None,
)
assert result.data is not None
assert len(result.data) == 1
assert result.data[0].url == image_url
def test_transform_response_multiple_images(self):
body = {
"output": {
"choices": [
{
"finish_reason": "stop",
"message": {
"role": "assistant",
"content": [{"image": "https://example.com/img1.png"}],
},
},
{
"finish_reason": "stop",
"message": {
"role": "assistant",
"content": [{"image": "https://example.com/img2.png"}],
},
},
]
},
"usage": {},
}
mock_resp = MagicMock(spec=httpx.Response)
mock_resp.status_code = 200
mock_resp.headers = {}
mock_resp.json.return_value = body
model_response = ImageResponse()
result = self.cfg.transform_image_generation_response(
model="qwen-image-2.0",
raw_response=mock_resp,
model_response=model_response,
logging_obj=MagicMock(),
request_data={},
optional_params={},
litellm_params={},
encoding=None,
)
assert len(result.data) == 2
assert result.data[0].url == "https://example.com/img1.png"
assert result.data[1].url == "https://example.com/img2.png"
def test_transform_response_raises_on_non_200_status(self):
mock_resp = MagicMock(spec=httpx.Response)
mock_resp.status_code = 400
mock_resp.headers = {}
mock_resp.text = '{"code":"InvalidParameter","message":"Size not supported"}'
mock_resp.json.return_value = {
"code": "InvalidParameter",
"message": "Size not supported",
}
with pytest.raises(Exception):
self.cfg.transform_image_generation_response(
model="qwen-image-2.0",
raw_response=mock_resp,
model_response=ImageResponse(),
logging_obj=MagicMock(),
request_data={},
optional_params={},
litellm_params={},
encoding=None,
)
def test_transform_response_raises_on_api_error_body(self):
mock_resp = MagicMock(spec=httpx.Response)
mock_resp.status_code = 200
mock_resp.headers = {}
mock_resp.json.return_value = {
"code": "InvalidParameter",
"message": "Size not supported",
}
with pytest.raises(Exception):
self.cfg.transform_image_generation_response(
model="qwen-image-2.0",
raw_response=mock_resp,
model_response=ImageResponse(),
logging_obj=MagicMock(),
request_data={},
optional_params={},
litellm_params={},
encoding=None,
)
# ---------------------------------------------------------------------------
# 5. OpenAI → DashScope parameter mapping
# ---------------------------------------------------------------------------
def test_map_openai_params_size_conversion(self):
mapped = self.cfg.map_openai_params(
non_default_params={"size": "1024x1024"},
optional_params={},
model="qwen-image-2.0",
drop_params=False,
)
assert mapped["size"] == "1024*1024"
def test_map_openai_params_n_to_image_count(self):
mapped = self.cfg.map_openai_params(
non_default_params={"n": 2},
optional_params={},
model="qwen-image-2.0",
drop_params=False,
)
assert mapped["image_count"] == 2
def test_map_openai_params_unknown_size_uses_asterisk(self):
mapped = self.cfg.map_openai_params(
non_default_params={"size": "768x768"},
optional_params={},
model="qwen-image-2.0",
drop_params=False,
)
assert mapped["size"] == "768*768"
@pytest.mark.parametrize(
"openai_size, expected",
[
("256x256", "256*256"),
("512x512", "512*512"),
("1024x1024", "1024*1024"),
("1792x1024", "1792*1024"),
("1024x1792", "1024*1792"),
("2048x2048", "2048*2048"),
],
)
def test_map_openai_params_size_table(self, openai_size: str, expected: str):
mapped = self.cfg.map_openai_params(
non_default_params={"size": openai_size},
optional_params={},
model="qwen-image-2.0",
drop_params=False,
)
assert mapped["size"] == expected
# ---------------------------------------------------------------------------
# 6. End-to-end flow via litellm.image_generation (HTTP mocked)
# ---------------------------------------------------------------------------
def test_litellm_image_generation_dashscope_end_to_end():
mock_response_body = {
"output": {
"choices": [
{
"finish_reason": "stop",
"message": {
"role": "assistant",
"content": [
{
"image": "https://dashscope-result.oss.aliyuncs.com/test.png"
}
],
},
}
]
},
"usage": {
"input_tokens": 0,
"output_tokens": 0,
"width": 1024,
"height": 1024,
"image_count": 1,
},
}
with patch(
"litellm.llms.custom_httpx.llm_http_handler.HTTPHandler.post"
) as mock_post:
mock_http_response = MagicMock()
mock_http_response.json.return_value = mock_response_body
mock_http_response.status_code = 200
mock_http_response.headers = {}
mock_post.return_value = mock_http_response
response = litellm.image_generation(
model="dashscope/qwen-image-2.0",
prompt="a puppy playing on green grass",
api_key="sk-test-key",
size="1024x1024",
)
assert response is not None
assert response.data is not None
assert len(response.data) == 1
assert (
response.data[0].url == "https://dashscope-result.oss.aliyuncs.com/test.png"
)
# Verify the HTTP call was made to the DashScope endpoint
call_args = mock_post.call_args
called_url = (
call_args[0][0] if call_args[0] else call_args.kwargs.get("url", "")
)
assert "dashscope" in called_url or "aliyuncs" in called_url
# Verify request body contains DashScope format
call_kwargs = call_args[1] if call_args[1] else {}
if "json" in call_kwargs:
body = call_kwargs["json"]
assert "input" in body
assert "messages" in body["input"]

View file

@ -243,6 +243,7 @@ export default function SpendLogsTable({
allTeams,
handleFilterChange,
handleFilterReset: handleFilterResetFromHook,
refetchWithFilters,
} = useLogFilterLogic({
logs: logsData,
accessToken,
@ -363,7 +364,14 @@ export default function SpendLogsTable({
// Add this function to handle manual refresh
const handleRefresh = () => {
logs.refetch();
if (hasBackendFilters) {
// When backend filters (e.g. Key Alias) are active the main TanStack Query
// is disabled and its params do not include filter values like key_alias.
// Route through the filter-aware refetch so all active filters are preserved.
refetchWithFilters();
} else {
logs.refetch();
}
};
const handleRowClick = (log: LogEntry) => {

View file

@ -71,6 +71,14 @@ export function useLogFilterLogic({
const [filters, setFilters] = useState<LogFilterState>(defaultFilters);
const [backendFilteredLogs, setBackendFilteredLogs] = useState<PaginatedResponse | null>(null);
const lastSearchTimestamp = useRef(0);
// Refs that always hold the latest filters and hasBackendFilters values.
// The sort/page/time effect below intentionally omits these from its dep array
// to avoid double-fetches when a filter changes; reading from refs instead of
// the closure prevents stale-closure bugs (e.g. the effect using a snapshot of
// filters taken before the user selected Key Alias).
const filtersRef = useRef(filters);
const hasBackendFiltersRef = useRef(false);
const performSearch = useCallback(
async (filters: LogFilterState, page = 1) => {
if (!accessToken) return;
@ -152,18 +160,25 @@ export function useLogFilterLogic({
[filters],
);
// Keep refs in sync on every render so the sort/page/time effect always reads
// the latest values without those values being in its dep array.
useEffect(() => {
filtersRef.current = filters;
hasBackendFiltersRef.current = hasBackendFilters;
}, [filters, hasBackendFilters]);
// Refetch when sort, page, or time range changes (backend filters use their own fetch, not the main query)
useEffect(() => {
if (hasBackendFilters && accessToken) {
if (hasBackendFiltersRef.current && accessToken) {
// Cancel any pending debounced search to prevent it from overwriting this page's results
debouncedSearch.cancel();
performSearch(filters, currentPage);
performSearch(filtersRef.current, currentPage);
}
// Intentionally omitted from deps:
// - `filters` / `debouncedSearch` / `performSearch`: filter changes are handled by
// handleFilterChange → debouncedSearch; adding them here would double-fetch on filter apply.
// - `hasBackendFilters` / `accessToken`: stable across sort/page/time changes; including them
// would cause spurious re-runs when the filter state first becomes active.
// filters / hasBackendFilters are read via refs — avoids stale-closure bugs
// when sort/page/time changes after a filter (e.g. Key Alias) was set.
// debouncedSearch / performSearch: filter changes go through handleFilterChange
// → debouncedSearch; adding them here would cause double-fetches on filter apply.
// accessToken: stable across sort/page/time changes.
// eslint-disable-next-line react-hooks/exhaustive-deps
}, [sortBy, sortOrder, currentPage, startTime, endTime, isCustomDate]);
@ -299,6 +314,20 @@ export function useLogFilterLogic({
setCurrentPage(1);
};
// Expose a filter-aware refetch so callers (e.g. the manual Fetch button) can
// refresh results while keeping all active backend filters intact. The plain
// `logs.refetch()` in the parent only re-runs the main TanStack Query, which
// does not carry key_alias or other backend-only filter params.
const refetchWithFilters = useCallback(
(page = currentPage) => {
if (hasBackendFilters && accessToken) {
debouncedSearch.cancel();
performSearch(filters, page);
}
},
[hasBackendFilters, accessToken, filters, currentPage, performSearch, debouncedSearch],
);
return {
filters,
filteredLogs,
@ -306,5 +335,6 @@ export function useLogFilterLogic({
allTeams,
handleFilterChange,
handleFilterReset,
refetchWithFilters,
};
}