mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-14 23:21:35 +00:00
Merge branch 'BerriAI:litellm_internal_staging' into feat/fireworks-think-tag-extraction
This commit is contained in:
commit
37bb6c1156
33 changed files with 2248 additions and 189 deletions
|
|
@ -2061,7 +2061,7 @@ assert isinstance(
|
|||
|
||||
## Media Resolution Control (Images & Videos)
|
||||
|
||||
For Gemini 3+ models, LiteLLM supports per-part media resolution control using OpenAI's `detail` parameter. This allows you to specify different resolution levels for individual images and videos in your request, whether using `image_url` or `file` content types.
|
||||
LiteLLM supports per-part media resolution control using OpenAI's `detail` parameter for all Gemini models. This allows you to specify different resolution levels for individual images and videos in your request, whether using `image_url` or `file` content types.
|
||||
|
||||
**Supported `detail` values:**
|
||||
- `"low"` - Maps to `media_resolution: "low"` (280 tokens for images, 70 tokens per frame for videos)
|
||||
|
|
@ -2146,12 +2146,12 @@ response = completion(
|
|||
</Tabs>
|
||||
|
||||
:::info
|
||||
**Per-Part Resolution:** Each image or video in your request can have its own `detail` setting, allowing mixed-resolution requests (e.g., a high-res chart alongside a low-res icon). This feature works with both `image_url` and `file` content types, and is only available for Gemini 3+ models.
|
||||
**Per-Part Resolution:** Each image or video in your request can have its own `detail` setting, allowing mixed-resolution requests (e.g., a high-res chart alongside a low-res icon). This feature works with both `image_url` and `file` content types across all Gemini models.
|
||||
:::
|
||||
|
||||
## Video Metadata Control
|
||||
|
||||
For Gemini 3+ models, LiteLLM supports fine-grained video processing control through the `video_metadata` field. This allows you to specify frame extraction rates and time ranges for video analysis.
|
||||
LiteLLM supports fine-grained video processing control through the `video_metadata` field for all Gemini models (1.x, 2.x, 3+). This allows you to specify frame extraction rates and time ranges for video analysis.
|
||||
|
||||
**Supported `video_metadata` parameters:**
|
||||
|
||||
|
|
@ -2168,8 +2168,11 @@ For Gemini 3+ models, LiteLLM supports fine-grained video processing control thr
|
|||
- `fps` remains unchanged
|
||||
:::
|
||||
|
||||
:::tip
|
||||
Video clipping (`start_offset`/`end_offset`) and frame rate control (`fps`) are supported by all Gemini models, but analysis quality is significantly higher with the **Gemini 2.5 series** (e.g., `gemini-2.5-flash`, `gemini-2.5-pro`).
|
||||
:::
|
||||
|
||||
:::warning
|
||||
- **Gemini 3+ Only:** This feature is only available for Gemini 3.0 and newer models
|
||||
- **Video Files Recommended:** While `video_metadata` is designed for video files, error handling for other media types is delegated to the Vertex AI API
|
||||
- **File Formats Supported:** Works with `gs://`, `https://`, and base64-encoded video files
|
||||
:::
|
||||
|
|
|
|||
|
|
@ -410,6 +410,7 @@ def image_generation( # noqa: PLR0915
|
|||
litellm.LlmProviders.RUNWAYML,
|
||||
litellm.LlmProviders.VERTEX_AI,
|
||||
litellm.LlmProviders.OPENROUTER,
|
||||
litellm.LlmProviders.DASHSCOPE,
|
||||
):
|
||||
if image_generation_config is None:
|
||||
raise ValueError(
|
||||
|
|
|
|||
|
|
@ -452,7 +452,14 @@ def update_messages_with_model_file_ids(
|
|||
for c in content:
|
||||
if c["type"] == "file":
|
||||
file_object = cast(ChatCompletionFileObject, c)
|
||||
file_object_file_field = file_object["file"]
|
||||
file_object_file_field = file_object.get("file")
|
||||
if not isinstance(file_object_file_field, dict):
|
||||
# Content block has `type: "file"` but not the
|
||||
# OpenAI Chat Completions shape (e.g. a LangChain
|
||||
# v1 standardized file block, or a provider-native
|
||||
# shape that also uses `type: "file"`). Nothing to
|
||||
# remap here, so skip instead of crashing.
|
||||
continue
|
||||
file_id = file_object_file_field.get("file_id")
|
||||
format = file_object_file_field.get(
|
||||
"format", get_format_from_file_id(file_id)
|
||||
|
|
@ -1060,7 +1067,12 @@ def get_file_ids_from_messages(messages: List[AllMessageValues]) -> List[str]:
|
|||
for c in content:
|
||||
if c["type"] == "file":
|
||||
file_object = cast(ChatCompletionFileObject, c)
|
||||
file_object_file_field = file_object["file"]
|
||||
file_object_file_field = file_object.get("file")
|
||||
if not isinstance(file_object_file_field, dict):
|
||||
# Content block has `type: "file"` but not the
|
||||
# OpenAI Chat Completions shape. No file_id to
|
||||
# extract, so skip instead of raising KeyError.
|
||||
continue
|
||||
file_id = file_object_file_field.get("file_id")
|
||||
if file_id:
|
||||
file_ids.append(file_id)
|
||||
|
|
|
|||
|
|
@ -106,6 +106,44 @@ class LiteLLMMessagesToCompletionTransformationHandler:
|
|||
updated_reasoning_effort["summary"] = effective_summary
|
||||
completion_kwargs["reasoning_effort"] = updated_reasoning_effort
|
||||
|
||||
@staticmethod
|
||||
def _normalize_reasoning_effort(
|
||||
completion_kwargs: Dict[str, Any],
|
||||
) -> None:
|
||||
"""
|
||||
Normalize reasoning_effort values based on target model capabilities.
|
||||
|
||||
Handles both string ("max") and dict ({"effort": "max", "summary": ...})
|
||||
formats. Uses model registry to check supports_xhigh/supports_minimal.
|
||||
"""
|
||||
from litellm.llms.anthropic.experimental_pass_through.utils import (
|
||||
normalize_reasoning_effort_value,
|
||||
)
|
||||
|
||||
reasoning_effort = completion_kwargs.get("reasoning_effort")
|
||||
if reasoning_effort is None:
|
||||
return
|
||||
|
||||
model = cast(str, completion_kwargs.get("model", ""))
|
||||
custom_llm_provider = completion_kwargs.get("custom_llm_provider")
|
||||
|
||||
if isinstance(reasoning_effort, str):
|
||||
normalized = normalize_reasoning_effort_value(
|
||||
reasoning_effort, model=model, custom_llm_provider=custom_llm_provider
|
||||
)
|
||||
if normalized != reasoning_effort:
|
||||
completion_kwargs["reasoning_effort"] = normalized
|
||||
elif isinstance(reasoning_effort, dict) and "effort" in reasoning_effort:
|
||||
effort = reasoning_effort["effort"]
|
||||
normalized = normalize_reasoning_effort_value(
|
||||
effort, model=model, custom_llm_provider=custom_llm_provider
|
||||
)
|
||||
if normalized != effort:
|
||||
completion_kwargs["reasoning_effort"] = {
|
||||
**reasoning_effort,
|
||||
"effort": normalized,
|
||||
}
|
||||
|
||||
@staticmethod
|
||||
def _prepare_completion_kwargs(
|
||||
*,
|
||||
|
|
@ -163,6 +201,12 @@ class LiteLLMMessagesToCompletionTransformationHandler:
|
|||
if output_format:
|
||||
request_data["output_format"] = output_format
|
||||
|
||||
# Extract output_config from extra_kwargs so the translator can use it
|
||||
# (e.g. output_config.effort for adaptive thinking → reasoning_effort)
|
||||
extra_kwargs = extra_kwargs or {}
|
||||
if "output_config" in extra_kwargs:
|
||||
request_data["output_config"] = extra_kwargs["output_config"]
|
||||
|
||||
(
|
||||
openai_request,
|
||||
tool_name_mapping,
|
||||
|
|
@ -202,6 +246,14 @@ class LiteLLMMessagesToCompletionTransformationHandler:
|
|||
):
|
||||
completion_kwargs[key] = value
|
||||
|
||||
# Normalize reasoning_effort based on model capabilities
|
||||
# (e.g. "max" → "xhigh"/"high", "minimal" → "low" if unsupported)
|
||||
# Must run BEFORE _route_openai_thinking, which prepends "responses/"
|
||||
# to the model name and would break get_model_info() lookups.
|
||||
LiteLLMMessagesToCompletionTransformationHandler._normalize_reasoning_effort(
|
||||
completion_kwargs
|
||||
)
|
||||
|
||||
LiteLLMMessagesToCompletionTransformationHandler._route_openai_thinking_to_responses_api_if_needed(
|
||||
completion_kwargs,
|
||||
thinking=thinking,
|
||||
|
|
|
|||
|
|
@ -317,6 +317,7 @@ class LiteLLMAnthropicMessagesAdapter:
|
|||
"tools",
|
||||
"thinking",
|
||||
"output_format",
|
||||
"output_config",
|
||||
]
|
||||
|
||||
def _is_web_search_tool(self, tool: Dict[str, Any]) -> bool:
|
||||
|
|
@ -694,6 +695,11 @@ class LiteLLMAnthropicMessagesAdapter:
|
|||
return "low"
|
||||
else:
|
||||
return "minimal"
|
||||
elif thinking_type == "adaptive":
|
||||
# Adaptive thinking: effort is controlled by output_config.effort,
|
||||
# not budget_tokens. Return a default; caller should override with
|
||||
# output_config.effort when available.
|
||||
return "medium"
|
||||
|
||||
return None
|
||||
|
||||
|
|
@ -776,6 +782,8 @@ class LiteLLMAnthropicMessagesAdapter:
|
|||
return ChatCompletionToolChoiceObjectParam(
|
||||
type="function", function=tc_function_param
|
||||
)
|
||||
elif tool_choice["type"] == "none":
|
||||
return "none"
|
||||
else:
|
||||
raise ValueError(
|
||||
"Incompatible tool choice param submitted - {}".format(tool_choice)
|
||||
|
|
@ -1041,6 +1049,12 @@ class LiteLLMAnthropicMessagesAdapter:
|
|||
if not reasoning_effort:
|
||||
return
|
||||
|
||||
# For adaptive thinking, override with output_config.effort if available
|
||||
if isinstance(thinking, dict) and thinking.get("type") == "adaptive":
|
||||
output_config = anthropic_message_request.get("output_config")
|
||||
if isinstance(output_config, dict) and output_config.get("effort"):
|
||||
reasoning_effort = output_config["effort"]
|
||||
|
||||
summary = thinking.get("summary") if isinstance(thinking, dict) else None
|
||||
auto_summary = is_reasoning_auto_summary_enabled()
|
||||
if summary:
|
||||
|
|
|
|||
|
|
@ -24,6 +24,8 @@ from litellm.types.llms.anthropic_messages.anthropic_response import (
|
|||
from litellm.types.router import GenericLiteLLMParams
|
||||
from litellm.utils import ProviderConfigManager, client
|
||||
|
||||
from ..utils import is_reasoning_auto_summary_enabled
|
||||
|
||||
from ..adapters.handler import LiteLLMMessagesToCompletionTransformationHandler
|
||||
from ..responses_adapters.handler import LiteLLMMessagesToResponsesAPIHandler
|
||||
from .interceptors import get_messages_interceptors
|
||||
|
|
@ -441,6 +443,17 @@ def anthropic_messages_handler(
|
|||
params=local_vars
|
||||
)
|
||||
)
|
||||
if is_reasoning_auto_summary_enabled():
|
||||
thinking_param = anthropic_messages_optional_request_params.get("thinking")
|
||||
if (
|
||||
isinstance(thinking_param, dict)
|
||||
and thinking_param.get("type") != "disabled"
|
||||
):
|
||||
anthropic_messages_optional_request_params["thinking"] = {
|
||||
**thinking_param,
|
||||
"display": "summarized",
|
||||
}
|
||||
|
||||
return base_llm_http_handler.anthropic_messages_handler(
|
||||
model=model,
|
||||
messages=messages,
|
||||
|
|
|
|||
|
|
@ -72,6 +72,23 @@ def _build_responses_kwargs(
|
|||
anthropic_request = AnthropicMessagesRequest(**request_data) # type: ignore[typeddict-item]
|
||||
responses_kwargs = _ADAPTER.translate_request(anthropic_request)
|
||||
|
||||
# Normalize reasoning effort based on model capabilities
|
||||
# (e.g. "max" → "xhigh"/"high", "minimal" → "low" if unsupported)
|
||||
reasoning = responses_kwargs.get("reasoning")
|
||||
if isinstance(reasoning, dict) and "effort" in reasoning:
|
||||
from litellm.llms.anthropic.experimental_pass_through.utils import (
|
||||
normalize_reasoning_effort_value,
|
||||
)
|
||||
|
||||
effort = reasoning["effort"]
|
||||
normalized = normalize_reasoning_effort_value(
|
||||
effort,
|
||||
model=model,
|
||||
custom_llm_provider=(extra_kwargs or {}).get("custom_llm_provider"),
|
||||
)
|
||||
if normalized != effort:
|
||||
responses_kwargs["reasoning"] = {**reasoning, "effort": normalized}
|
||||
|
||||
if stream:
|
||||
responses_kwargs["stream"] = True
|
||||
|
||||
|
|
|
|||
|
|
@ -251,25 +251,41 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
|
|||
|
||||
@staticmethod
|
||||
def translate_thinking_to_reasoning(
|
||||
thinking: Dict[str, Any]
|
||||
thinking: Dict[str, Any],
|
||||
output_config: Optional[Dict[str, Any]] = None,
|
||||
) -> Optional[Dict[str, Any]]:
|
||||
"""
|
||||
Convert Anthropic thinking param to Responses API reasoning param.
|
||||
|
||||
thinking.budget_tokens maps to reasoning effort:
|
||||
>= 10000 -> high, >= 5000 -> medium, >= 2000 -> low, < 2000 -> minimal
|
||||
|
||||
For adaptive thinking, uses output_config.effort if available,
|
||||
otherwise defaults to medium.
|
||||
"""
|
||||
if not isinstance(thinking, dict) or thinking.get("type") != "enabled":
|
||||
if not isinstance(thinking, dict):
|
||||
return None
|
||||
budget = thinking.get("budget_tokens", 0)
|
||||
if budget >= 10000:
|
||||
effort = "high"
|
||||
elif budget >= 5000:
|
||||
|
||||
thinking_type = thinking.get("type")
|
||||
|
||||
if thinking_type == "adaptive":
|
||||
# Use output_config.effort if available
|
||||
effort = "medium"
|
||||
elif budget >= 2000:
|
||||
effort = "low"
|
||||
if isinstance(output_config, dict) and output_config.get("effort"):
|
||||
effort = output_config["effort"]
|
||||
elif thinking_type == "enabled":
|
||||
budget = thinking.get("budget_tokens", 0)
|
||||
if budget >= 10000:
|
||||
effort = "high"
|
||||
elif budget >= 5000:
|
||||
effort = "medium"
|
||||
elif budget >= 2000:
|
||||
effort = "low"
|
||||
else:
|
||||
effort = "minimal"
|
||||
else:
|
||||
effort = "minimal"
|
||||
return None
|
||||
|
||||
auto_summary = is_reasoning_auto_summary_enabled()
|
||||
result: Dict[str, Any] = {"effort": effort}
|
||||
summary = thinking.get("summary")
|
||||
|
|
@ -346,7 +362,11 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
|
|||
# thinking -> reasoning
|
||||
thinking = anthropic_request.get("thinking")
|
||||
if isinstance(thinking, dict):
|
||||
reasoning = self.translate_thinking_to_reasoning(thinking)
|
||||
output_config = anthropic_request.get("output_config")
|
||||
reasoning = self.translate_thinking_to_reasoning(
|
||||
thinking,
|
||||
output_config=cast(Optional[Dict[str, Any]], output_config),
|
||||
)
|
||||
if reasoning:
|
||||
responses_kwargs["reasoning"] = reasoning
|
||||
|
||||
|
|
|
|||
|
|
@ -1,6 +1,8 @@
|
|||
import os
|
||||
from typing import Optional
|
||||
|
||||
import litellm
|
||||
from litellm.types.utils import ModelInfo
|
||||
|
||||
|
||||
def is_reasoning_auto_summary_enabled() -> bool:
|
||||
|
|
@ -9,3 +11,47 @@ def is_reasoning_auto_summary_enabled() -> bool:
|
|||
litellm.reasoning_auto_summary
|
||||
or os.getenv("LITELLM_REASONING_AUTO_SUMMARY", "false").lower() == "true"
|
||||
)
|
||||
|
||||
|
||||
def normalize_reasoning_effort_value(
|
||||
effort: str,
|
||||
model: str,
|
||||
custom_llm_provider: Optional[str] = None,
|
||||
) -> str:
|
||||
"""
|
||||
Normalize a reasoning effort value based on model capabilities.
|
||||
|
||||
Degradation chains:
|
||||
- "max" → max / xhigh / high
|
||||
- "xhigh" → xhigh / high
|
||||
- "minimal" → minimal / low
|
||||
- other values pass through unchanged
|
||||
"""
|
||||
if effort not in ("max", "xhigh", "minimal"):
|
||||
return effort
|
||||
|
||||
from litellm.utils import get_model_info
|
||||
|
||||
model_info: Optional[ModelInfo] = None
|
||||
try:
|
||||
model_info = get_model_info(
|
||||
model=model, custom_llm_provider=custom_llm_provider
|
||||
)
|
||||
except Exception:
|
||||
model_info = None
|
||||
|
||||
if effort == "max":
|
||||
if model_info and model_info.get("supports_max_reasoning_effort"):
|
||||
return "max"
|
||||
if model_info and model_info.get("supports_xhigh_reasoning_effort"):
|
||||
return "xhigh"
|
||||
return "high"
|
||||
elif effort == "xhigh":
|
||||
if model_info and model_info.get("supports_xhigh_reasoning_effort"):
|
||||
return "xhigh"
|
||||
return "high"
|
||||
elif effort == "minimal":
|
||||
if model_info and model_info.get("supports_minimal_reasoning_effort"):
|
||||
return "minimal"
|
||||
return "low"
|
||||
return "medium"
|
||||
|
|
|
|||
|
|
@ -40,9 +40,22 @@ class AzureOpenAIGPT5Config(AzureOpenAIConfig, OpenAIGPT5Config):
|
|||
Accepts both explicit gpt-5 model names and the ``gpt5_series/`` prefix
|
||||
used for manual routing.
|
||||
"""
|
||||
# gpt-5-chat* is a chat model and shouldn't go through GPT-5 reasoning restrictions.
|
||||
# The gpt-5-chat* family (gpt-5-chat, gpt-5-chat-latest, gpt-5-chat-2025-08-07,
|
||||
# …) are regular chat models: they support temperature and tool_choice but NOT
|
||||
# reasoning_effort. They must NOT be routed through the GPT-5 reasoning path.
|
||||
#
|
||||
# Versioned chat models such as gpt-5.3-chat and gpt-5.1-chat ARE reasoning
|
||||
# models and must stay on the GPT-5 path. The distinguishing feature is that
|
||||
# the gpt-5-chat family has a literal "-chat" immediately after "gpt-5"
|
||||
# (i.e. "gpt-5-chat…"), while versioned chat models interpose a minor version
|
||||
# number (i.e. "gpt-5.<digit>-chat").
|
||||
#
|
||||
# Using a startswith("gpt-5-chat") prefix check on the normalized name (rather
|
||||
# than a substring check) makes this boundary explicit and avoids any ambiguity
|
||||
# if future model names coincidentally contain "gpt-5-chat" as an interior run.
|
||||
_normalized = model.split("/")[-1] # strip provider prefix, e.g. "azure/"
|
||||
return (
|
||||
"gpt-5" in model and "gpt-5-chat" not in model
|
||||
"gpt-5" in model and not _normalized.startswith("gpt-5-chat")
|
||||
) or "gpt5_series" in model
|
||||
|
||||
def get_supported_openai_params(self, model: str) -> List[str]:
|
||||
|
|
|
|||
11
litellm/llms/dashscope/image_generation/__init__.py
Normal file
11
litellm/llms/dashscope/image_generation/__init__.py
Normal file
|
|
@ -0,0 +1,11 @@
|
|||
from litellm.llms.base_llm.image_generation.transformation import (
|
||||
BaseImageGenerationConfig,
|
||||
)
|
||||
|
||||
from .transformation import DashScopeImageGenerationConfig
|
||||
|
||||
__all__ = ["DashScopeImageGenerationConfig"]
|
||||
|
||||
|
||||
def get_dashscope_image_generation_config(model: str) -> BaseImageGenerationConfig:
|
||||
return DashScopeImageGenerationConfig()
|
||||
204
litellm/llms/dashscope/image_generation/transformation.py
Normal file
204
litellm/llms/dashscope/image_generation/transformation.py
Normal file
|
|
@ -0,0 +1,204 @@
|
|||
"""
|
||||
DashScope Image Generation Configuration
|
||||
|
||||
Handles transformation between OpenAI-compatible format and DashScope multimodal-generation API.
|
||||
|
||||
API endpoint: POST https://dashscope-intl.aliyuncs.com/api/v1/services/aigc/multimodal-generation/generation
|
||||
|
||||
Request format:
|
||||
{
|
||||
"model": "qwen-image-2.0-pro",
|
||||
"input": {
|
||||
"messages": [{"role": "user", "content": [{"text": "<prompt>"}]}]
|
||||
},
|
||||
"parameters": {"size": "1024*1024", ...}
|
||||
}
|
||||
|
||||
Response format:
|
||||
{
|
||||
"output": {
|
||||
"choices": [{"message": {"content": [{"image": "<url>"}]}}]
|
||||
},
|
||||
"usage": {"input_tokens": 0, "output_tokens": 0, "width": 1024, "height": 1024, "image_count": 1}
|
||||
}
|
||||
"""
|
||||
|
||||
from typing import TYPE_CHECKING, Any, List, Optional
|
||||
|
||||
import httpx
|
||||
|
||||
from litellm.llms.base_llm.image_generation.transformation import (
|
||||
BaseImageGenerationConfig,
|
||||
)
|
||||
from litellm.secret_managers.main import get_secret_str
|
||||
from litellm.types.llms.openai import (
|
||||
AllMessageValues,
|
||||
OpenAIImageGenerationOptionalParams,
|
||||
)
|
||||
from litellm.types.utils import ImageObject, ImageResponse
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as _LiteLLMLoggingObj
|
||||
|
||||
LiteLLMLoggingObj = _LiteLLMLoggingObj
|
||||
else:
|
||||
LiteLLMLoggingObj = Any
|
||||
|
||||
DEFAULT_API_BASE = "https://dashscope-intl.aliyuncs.com/api/v1/services/aigc/multimodal-generation/generation"
|
||||
|
||||
# Maps OpenAI size strings (WxH) to DashScope size strings (W*H)
|
||||
OPENAI_TO_DASHSCOPE_SIZE: dict = {
|
||||
"256x256": "256*256",
|
||||
"512x512": "512*512",
|
||||
"1024x1024": "1024*1024",
|
||||
"1792x1024": "1792*1024",
|
||||
"1024x1792": "1024*1792",
|
||||
"2048x2048": "2048*2048",
|
||||
}
|
||||
|
||||
|
||||
class DashScopeImageGenerationConfig(BaseImageGenerationConfig):
|
||||
"""
|
||||
Configuration for DashScope image generation (qwen-image-2.0, qwen-image-2.0-pro).
|
||||
"""
|
||||
|
||||
def get_supported_openai_params(
|
||||
self, model: str
|
||||
) -> List[OpenAIImageGenerationOptionalParams]:
|
||||
return ["n", "size"]
|
||||
|
||||
def map_openai_params(
|
||||
self,
|
||||
non_default_params: dict,
|
||||
optional_params: dict,
|
||||
model: str,
|
||||
drop_params: bool,
|
||||
) -> dict:
|
||||
supported_params = self.get_supported_openai_params(model)
|
||||
mapped: dict = {}
|
||||
for k, v in non_default_params.items():
|
||||
if k in optional_params:
|
||||
continue
|
||||
if k not in supported_params:
|
||||
continue
|
||||
if k == "size":
|
||||
# Convert "WxH" → "W*H"
|
||||
mapped["size"] = OPENAI_TO_DASHSCOPE_SIZE.get(v, v.replace("x", "*"))
|
||||
elif k == "n":
|
||||
mapped["image_count"] = v
|
||||
return mapped
|
||||
|
||||
def get_complete_url(
|
||||
self,
|
||||
api_base: Optional[str],
|
||||
api_key: Optional[str],
|
||||
model: str,
|
||||
optional_params: dict,
|
||||
litellm_params: dict,
|
||||
stream: Optional[bool] = None,
|
||||
) -> str:
|
||||
return (
|
||||
api_base or get_secret_str("DASHSCOPE_API_BASE_IMAGE") or DEFAULT_API_BASE
|
||||
)
|
||||
|
||||
def validate_environment(
|
||||
self,
|
||||
headers: dict,
|
||||
model: str,
|
||||
messages: List[AllMessageValues],
|
||||
optional_params: dict,
|
||||
litellm_params: dict,
|
||||
api_key: Optional[str] = None,
|
||||
api_base: Optional[str] = None,
|
||||
) -> dict:
|
||||
final_api_key = api_key or get_secret_str("DASHSCOPE_API_KEY")
|
||||
if not final_api_key:
|
||||
raise ValueError("DASHSCOPE_API_KEY is not set")
|
||||
headers["Authorization"] = f"Bearer {final_api_key}"
|
||||
headers["Content-Type"] = "application/json"
|
||||
return headers
|
||||
|
||||
def transform_image_generation_request(
|
||||
self,
|
||||
model: str,
|
||||
prompt: str,
|
||||
optional_params: dict,
|
||||
litellm_params: dict,
|
||||
headers: dict,
|
||||
) -> dict:
|
||||
"""
|
||||
Transform OpenAI-style image generation request to DashScope multimodal-generation format.
|
||||
"""
|
||||
parameters: dict = {}
|
||||
for k, v in optional_params.items():
|
||||
parameters[k] = v
|
||||
|
||||
return {
|
||||
"model": model,
|
||||
"input": {
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": [{"text": prompt}],
|
||||
}
|
||||
]
|
||||
},
|
||||
"parameters": parameters,
|
||||
}
|
||||
|
||||
def transform_image_generation_response(
|
||||
self,
|
||||
model: str,
|
||||
raw_response: httpx.Response,
|
||||
model_response: ImageResponse,
|
||||
logging_obj: LiteLLMLoggingObj,
|
||||
request_data: dict,
|
||||
optional_params: dict,
|
||||
litellm_params: dict,
|
||||
encoding: Any,
|
||||
api_key: Optional[str] = None,
|
||||
json_mode: Optional[bool] = None,
|
||||
) -> ImageResponse:
|
||||
"""
|
||||
Transform DashScope response to litellm ImageResponse.
|
||||
|
||||
DashScope response: output.choices[0].message.content[0].image
|
||||
OpenAI response: data[0].url
|
||||
"""
|
||||
if raw_response.status_code != 200:
|
||||
raise self.get_error_class(
|
||||
error_message=raw_response.text,
|
||||
status_code=raw_response.status_code,
|
||||
headers=raw_response.headers,
|
||||
)
|
||||
|
||||
try:
|
||||
response_data = raw_response.json()
|
||||
except Exception as e:
|
||||
raise self.get_error_class(
|
||||
error_message=f"Failed to parse DashScope image generation response: {e}",
|
||||
status_code=raw_response.status_code,
|
||||
headers=raw_response.headers,
|
||||
)
|
||||
|
||||
# DashScope can return API-level errors in a 200 response body.
|
||||
# Example: {"code": "InvalidParameter", "message": "Size not supported"}
|
||||
if "code" in response_data and "output" not in response_data:
|
||||
raise self.get_error_class(
|
||||
error_message=str(response_data.get("message", response_data)),
|
||||
status_code=raw_response.status_code,
|
||||
headers=raw_response.headers,
|
||||
)
|
||||
|
||||
if not model_response.data:
|
||||
model_response.data = []
|
||||
|
||||
choices = response_data.get("output", {}).get("choices", [])
|
||||
for choice in choices:
|
||||
content_list = choice.get("message", {}).get("content", [])
|
||||
for content_item in content_list:
|
||||
image_url = content_item.get("image")
|
||||
if image_url:
|
||||
model_response.data.append(ImageObject(url=image_url))
|
||||
|
||||
return model_response
|
||||
|
|
@ -53,9 +53,21 @@ class OpenAIGPT5Config(OpenAIGPTConfig):
|
|||
|
||||
@classmethod
|
||||
def is_model_gpt_5_model(cls, model: str) -> bool:
|
||||
# gpt-5-chat* behaves like a regular chat model (supports temperature, etc.)
|
||||
# Don't route it through GPT-5 reasoning-specific parameter restrictions.
|
||||
return "gpt-5" in model and "gpt-5-chat" not in model
|
||||
# The gpt-5-chat* family (gpt-5-chat, gpt-5-chat-latest, gpt-5-chat-2025-08-07,
|
||||
# …) are regular chat models: they support temperature and tool_choice but NOT
|
||||
# reasoning_effort. They must NOT be routed through the GPT-5 reasoning path.
|
||||
#
|
||||
# Versioned chat models such as gpt-5.3-chat and gpt-5.1-chat ARE reasoning
|
||||
# models and must stay on the GPT-5 path. The distinguishing feature is that
|
||||
# the gpt-5-chat family has a literal "-chat" immediately after "gpt-5"
|
||||
# (i.e. "gpt-5-chat…"), while versioned chat models interpose a minor version
|
||||
# number (i.e. "gpt-5.<digit>-chat").
|
||||
#
|
||||
# Using a startswith("gpt-5-chat") prefix check on the normalized name (rather
|
||||
# than a substring check) makes this boundary explicit and avoids any ambiguity
|
||||
# if future model names coincidentally contain "gpt-5-chat" as an interior run.
|
||||
_normalized = model.split("/")[-1] # strip provider prefix, e.g. "openai/"
|
||||
return "gpt-5" in model and not _normalized.startswith("gpt-5-chat")
|
||||
|
||||
@classmethod
|
||||
def is_model_gpt_5_search_model(cls, model: str) -> bool:
|
||||
|
|
|
|||
|
|
@ -8,9 +8,8 @@ More information on our website: https://endpoints.ai.cloud.ovh.net
|
|||
from typing import Optional, Union, List
|
||||
|
||||
import httpx
|
||||
from litellm.utils import ModelResponseStream, _get_model_info_helper
|
||||
from litellm.utils import ModelResponseStream
|
||||
from litellm.llms.openai.chat.gpt_transformation import OpenAIGPTConfig
|
||||
from litellm._logging import verbose_logger
|
||||
from litellm.llms.ovhcloud.utils import OVHCloudException
|
||||
from litellm.llms.base_llm.base_model_iterator import BaseModelResponseIterator
|
||||
from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
||||
|
|
@ -22,34 +21,6 @@ class OVHCloudChatConfig(OpenAIGPTConfig):
|
|||
def custom_llm_provider(self) -> Optional[str]:
|
||||
return "ovhcloud"
|
||||
|
||||
def get_supported_openai_params(self, model: str) -> list:
|
||||
"""
|
||||
Details about function calling support can be found here:
|
||||
https://help.ovhcloud.com/csm/en-gb-public-cloud-ai-endpoints-function-calling?id=kb_article_view&sysparm_article=KB0071907
|
||||
"""
|
||||
supports_function_calling: Optional[bool] = None
|
||||
try:
|
||||
model_info = _get_model_info_helper(model, custom_llm_provider="ovhcloud")
|
||||
supports_function_calling = model_info.get(
|
||||
"supports_function_calling", None
|
||||
)
|
||||
if supports_function_calling is None:
|
||||
supports_function_calling = False
|
||||
except Exception as e:
|
||||
verbose_logger.debug(f"Error getting supported OpenAI params: {e}")
|
||||
supports_function_calling = False
|
||||
|
||||
optional_params = super().get_supported_openai_params(model)
|
||||
if supports_function_calling is not True:
|
||||
verbose_logger.debug(
|
||||
"You can see our models supporting function_calling in our catalog: https://endpoints.ai.cloud.ovh.net/catalog "
|
||||
)
|
||||
optional_params.remove("tools")
|
||||
optional_params.remove("tool_choice")
|
||||
optional_params.remove("function_call")
|
||||
optional_params.remove("response_format")
|
||||
return optional_params
|
||||
|
||||
def get_complete_url(
|
||||
self,
|
||||
api_base: Optional[str],
|
||||
|
|
|
|||
|
|
@ -132,26 +132,28 @@ def _extract_max_media_resolution_from_messages(
|
|||
return max_resolution
|
||||
|
||||
|
||||
def _apply_gemini_3_metadata(
|
||||
def _apply_gemini_metadata(
|
||||
part: PartType,
|
||||
model: Optional[str],
|
||||
media_resolution_enum: Optional[Dict[str, str]],
|
||||
video_metadata: Optional[Dict[str, Any]],
|
||||
) -> PartType:
|
||||
"""
|
||||
Apply the unique media_resolution and video_metadata parameters of Gemini 3+
|
||||
Apply media_resolution and video_metadata parameters to a Gemini part.
|
||||
|
||||
- Per-part media_resolution: Gemini 3+ only (2.x uses generation_config global).
|
||||
- video_metadata (fps, startOffset, endOffset): all Gemini models (1.x, 2.x, 3+).
|
||||
"""
|
||||
if model is None:
|
||||
return part
|
||||
|
||||
from .vertex_and_google_ai_studio_gemini import VertexGeminiConfig
|
||||
|
||||
if not VertexGeminiConfig._is_gemini_3_or_newer(model):
|
||||
return part
|
||||
|
||||
part_dict = dict(part)
|
||||
|
||||
if media_resolution_enum is not None:
|
||||
if media_resolution_enum is not None and VertexGeminiConfig._is_gemini_3_or_newer(
|
||||
model
|
||||
):
|
||||
part_dict["media_resolution"] = media_resolution_enum
|
||||
|
||||
if video_metadata is not None:
|
||||
|
|
@ -206,7 +208,7 @@ def _process_gemini_media(
|
|||
mime_type = format
|
||||
file_data = FileDataType(mime_type=mime_type, file_uri=image_url)
|
||||
part: PartType = {"file_data": file_data}
|
||||
return _apply_gemini_3_metadata(
|
||||
return _apply_gemini_metadata(
|
||||
part, model, media_resolution_enum, video_metadata
|
||||
)
|
||||
elif (
|
||||
|
|
@ -216,14 +218,14 @@ def _process_gemini_media(
|
|||
):
|
||||
file_data = FileDataType(mime_type=image_type, file_uri=image_url)
|
||||
part = {"file_data": file_data}
|
||||
return _apply_gemini_3_metadata(
|
||||
return _apply_gemini_metadata(
|
||||
part, model, media_resolution_enum, video_metadata
|
||||
)
|
||||
elif "http://" in image_url or "https://" in image_url or "base64" in image_url:
|
||||
image = convert_to_anthropic_image_obj(image_url, format=format)
|
||||
_blob: BlobType = {"data": image["data"], "mime_type": image["media_type"]}
|
||||
part = {"inline_data": cast(BlobType, _blob)}
|
||||
return _apply_gemini_3_metadata(
|
||||
return _apply_gemini_metadata(
|
||||
part, model, media_resolution_enum, video_metadata
|
||||
)
|
||||
raise Exception("Invalid image received - {}".format(image_url))
|
||||
|
|
@ -733,9 +735,9 @@ def _transform_request_body( # noqa: PLR0915
|
|||
**filtered_params
|
||||
)
|
||||
|
||||
# For Gemini 2.x models, add media_resolution to generation_config (global)
|
||||
# Gemini 3+ supports per-part media_resolution, but 2.x only supports global
|
||||
# Gemini 1.x does not support mediaResolution at all
|
||||
# For Gemini 2.x models, also add media_resolution to generation_config (global)
|
||||
# as a fallback, since some 2.x versions may not support per-part media_resolution.
|
||||
# Gemini 1.x does not support mediaResolution at all.
|
||||
if "gemini-2" in model:
|
||||
max_media_resolution = _extract_max_media_resolution_from_messages(messages)
|
||||
if max_media_resolution:
|
||||
|
|
|
|||
|
|
@ -1006,7 +1006,8 @@
|
|||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true,
|
||||
"supports_max_reasoning_effort": true
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"global.anthropic.claude-opus-4-6-v1": {
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
|
|
@ -1034,7 +1035,8 @@
|
|||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true,
|
||||
"supports_max_reasoning_effort": true
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"us.anthropic.claude-opus-4-6-v1": {
|
||||
"cache_creation_input_token_cost": 6.875e-06,
|
||||
|
|
@ -1062,7 +1064,8 @@
|
|||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true,
|
||||
"supports_max_reasoning_effort": true
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"eu.anthropic.claude-opus-4-6-v1": {
|
||||
"cache_creation_input_token_cost": 6.875e-06,
|
||||
|
|
@ -1090,7 +1093,8 @@
|
|||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true,
|
||||
"supports_max_reasoning_effort": true
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"au.anthropic.claude-opus-4-6-v1": {
|
||||
"cache_creation_input_token_cost": 6.875e-06,
|
||||
|
|
@ -1118,7 +1122,8 @@
|
|||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true,
|
||||
"supports_max_reasoning_effort": true
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"anthropic.claude-opus-4-7": {
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
|
|
@ -1146,7 +1151,9 @@
|
|||
"supports_vision": true,
|
||||
"supports_xhigh_reasoning_effort": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true
|
||||
"supports_native_structured_output": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"global.anthropic.claude-opus-4-7": {
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
|
|
@ -1174,7 +1181,9 @@
|
|||
"supports_vision": true,
|
||||
"supports_xhigh_reasoning_effort": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true
|
||||
"supports_native_structured_output": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"us.anthropic.claude-opus-4-7": {
|
||||
"cache_creation_input_token_cost": 6.875e-06,
|
||||
|
|
@ -1202,7 +1211,9 @@
|
|||
"supports_vision": true,
|
||||
"supports_xhigh_reasoning_effort": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true
|
||||
"supports_native_structured_output": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"eu.anthropic.claude-opus-4-7": {
|
||||
"cache_creation_input_token_cost": 6.875e-06,
|
||||
|
|
@ -1230,7 +1241,9 @@
|
|||
"supports_vision": true,
|
||||
"supports_xhigh_reasoning_effort": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true
|
||||
"supports_native_structured_output": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"au.anthropic.claude-opus-4-7": {
|
||||
"cache_creation_input_token_cost": 6.875e-06,
|
||||
|
|
@ -1258,7 +1271,9 @@
|
|||
"supports_vision": true,
|
||||
"supports_xhigh_reasoning_effort": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true
|
||||
"supports_native_structured_output": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"anthropic.claude-sonnet-4-6": {
|
||||
"cache_creation_input_token_cost": 3.75e-06,
|
||||
|
|
@ -1285,7 +1300,8 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true
|
||||
"supports_native_structured_output": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"global.anthropic.claude-sonnet-4-6": {
|
||||
"cache_creation_input_token_cost": 3.75e-06,
|
||||
|
|
@ -1312,7 +1328,8 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true
|
||||
"supports_native_structured_output": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"us.anthropic.claude-sonnet-4-6": {
|
||||
"cache_creation_input_token_cost": 4.125e-06,
|
||||
|
|
@ -1339,7 +1356,8 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true
|
||||
"supports_native_structured_output": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"eu.anthropic.claude-sonnet-4-6": {
|
||||
"cache_creation_input_token_cost": 4.125e-06,
|
||||
|
|
@ -1366,7 +1384,8 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true
|
||||
"supports_native_structured_output": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"au.anthropic.claude-sonnet-4-6": {
|
||||
"cache_creation_input_token_cost": 4.125e-06,
|
||||
|
|
@ -1393,7 +1412,8 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true
|
||||
"supports_native_structured_output": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"anthropic.claude-sonnet-4-20250514-v1:0": {
|
||||
"cache_creation_input_token_cost": 3.75e-06,
|
||||
|
|
@ -1911,7 +1931,8 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 159,
|
||||
"supports_max_reasoning_effort": true
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"azure_ai/claude-opus-4-7": {
|
||||
"input_cost_per_token": 5e-06,
|
||||
|
|
@ -1939,7 +1960,9 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"supports_xhigh_reasoning_effort": true,
|
||||
"tool_use_system_prompt_tokens": 159
|
||||
"tool_use_system_prompt_tokens": 159,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"azure_ai/claude-opus-4-1": {
|
||||
"cache_creation_input_token_cost": 1.875e-05,
|
||||
|
|
@ -2003,7 +2026,8 @@
|
|||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"azure/computer-use-preview": {
|
||||
"input_cost_per_token": 3e-06,
|
||||
|
|
@ -8909,7 +8933,8 @@
|
|||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"claude-sonnet-4-5-20250929-v1:0": {
|
||||
"cache_creation_input_token_cost": 3.75e-06,
|
||||
|
|
@ -9103,7 +9128,8 @@
|
|||
"us": 1.1,
|
||||
"fast": 6.0
|
||||
},
|
||||
"supports_max_reasoning_effort": true
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"claude-opus-4-6-20260205": {
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
|
|
@ -9135,7 +9161,8 @@
|
|||
"us": 1.1,
|
||||
"fast": 6.0
|
||||
},
|
||||
"supports_max_reasoning_effort": true
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"claude-opus-4-7": {
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
|
|
@ -9167,7 +9194,9 @@
|
|||
"provider_specific_entry": {
|
||||
"us": 1.1,
|
||||
"fast": 6.0
|
||||
}
|
||||
},
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"claude-opus-4-7-20260416": {
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
|
|
@ -9199,7 +9228,9 @@
|
|||
"provider_specific_entry": {
|
||||
"us": 1.1,
|
||||
"fast": 6.0
|
||||
}
|
||||
},
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"claude-sonnet-4-20250514": {
|
||||
"deprecation_date": "2026-05-14",
|
||||
|
|
@ -10352,6 +10383,22 @@
|
|||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"dashscope/qwen-image-2.0": {
|
||||
"litellm_provider": "dashscope",
|
||||
"mode": "image_generation",
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models",
|
||||
"supported_endpoints": [
|
||||
"/v1/images/generations"
|
||||
]
|
||||
},
|
||||
"dashscope/qwen-image-2.0-pro": {
|
||||
"litellm_provider": "dashscope",
|
||||
"mode": "image_generation",
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models",
|
||||
"supported_endpoints": [
|
||||
"/v1/images/generations"
|
||||
]
|
||||
},
|
||||
"databricks/databricks-bge-large-en": {
|
||||
"input_cost_per_token": 1.0003e-07,
|
||||
"input_dbu_cost_per_token": 1.429e-06,
|
||||
|
|
@ -25068,7 +25115,8 @@
|
|||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 159
|
||||
"tool_use_system_prompt_tokens": 159,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"openrouter/anthropic/claude-opus-4.5": {
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
|
|
@ -25106,7 +25154,8 @@
|
|||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"openrouter/anthropic/claude-sonnet-4.5": {
|
||||
"input_cost_per_image": 0.0048,
|
||||
|
|
@ -30134,7 +30183,8 @@
|
|||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true
|
||||
"supports_vision": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"vercel_ai_gateway/anthropic/claude-sonnet-4": {
|
||||
"cache_creation_input_token_cost": 3.75e-06,
|
||||
|
|
@ -31361,7 +31411,8 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_max_reasoning_effort": true
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"vertex_ai/claude-opus-4-6@default": {
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
|
|
@ -31388,7 +31439,8 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_max_reasoning_effort": true
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"vertex_ai/claude-opus-4-7": {
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
|
|
@ -31415,7 +31467,9 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"supports_xhigh_reasoning_effort": true,
|
||||
"tool_use_system_prompt_tokens": 346
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"vertex_ai/claude-opus-4-7@default": {
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
|
|
@ -31442,7 +31496,9 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"supports_xhigh_reasoning_effort": true,
|
||||
"tool_use_system_prompt_tokens": 346
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"vertex_ai/claude-sonnet-4-5": {
|
||||
"cache_creation_input_token_cost": 3.75e-06,
|
||||
|
|
@ -31494,7 +31550,8 @@
|
|||
"search_context_size_high": 0.01,
|
||||
"search_context_size_low": 0.01,
|
||||
"search_context_size_medium": 0.01
|
||||
}
|
||||
},
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"vertex_ai/claude-sonnet-4-5@20250929": {
|
||||
"cache_creation_input_token_cost": 3.75e-06,
|
||||
|
|
@ -38361,7 +38418,8 @@
|
|||
"search_context_size_high": 0.01,
|
||||
"search_context_size_low": 0.01,
|
||||
"search_context_size_medium": 0.01
|
||||
}
|
||||
},
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"duckduckgo/search": {
|
||||
"litellm_provider": "duckduckgo",
|
||||
|
|
|
|||
|
|
@ -7,6 +7,7 @@ Filters MCP tools semantically for /chat/completions and /responses endpoints.
|
|||
from typing import TYPE_CHECKING, Any, Dict, List, Optional
|
||||
|
||||
from litellm._logging import verbose_logger
|
||||
from litellm.proxy._experimental.mcp_server.utils import MCP_TOOL_PREFIX_SEPARATOR
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from semantic_router.routers import SemanticRouter
|
||||
|
|
@ -214,20 +215,89 @@ class SemanticMCPToolFilter:
|
|||
|
||||
return []
|
||||
|
||||
@staticmethod
|
||||
def _name_matches_canonical(client_name: str, canonical: str) -> bool:
|
||||
"""
|
||||
Return True if a client-side tool name refers to the given canonical
|
||||
MCP tool name.
|
||||
|
||||
MCP clients (e.g. opencode) commonly wrap the proxy's canonical tool
|
||||
name with an additive namespace prefix of their own
|
||||
(``<client_alias><sep><canonical>``). The prefix can use either a
|
||||
dash or an underscore as separator regardless of what
|
||||
``MCP_TOOL_PREFIX_SEPARATOR`` is set to on the proxy, because the
|
||||
client doesn't know the proxy's separator.
|
||||
|
||||
The match is anchored: ``canonical`` must form the complete suffix
|
||||
of ``client_name`` and be preceded by a separator character, so
|
||||
``rain_gear`` does not match canonical ``ear``.
|
||||
|
||||
Suffix matching is additionally gated on ``canonical`` itself
|
||||
containing ``MCP_TOOL_PREFIX_SEPARATOR``. Server-registered MCP
|
||||
tools are always emitted as
|
||||
``<server_name><MCP_TOOL_PREFIX_SEPARATOR><tool_name>`` (see
|
||||
``add_server_prefix_to_name``), so a canonical without the
|
||||
separator is not a namespaced MCP tool and falling back to
|
||||
suffix matching would spuriously collide with unrelated local
|
||||
user functions whose names end in the same characters.
|
||||
"""
|
||||
if client_name == canonical:
|
||||
return True
|
||||
if MCP_TOOL_PREFIX_SEPARATOR not in canonical:
|
||||
return False
|
||||
if len(client_name) <= len(canonical):
|
||||
return False
|
||||
if not client_name.endswith(canonical):
|
||||
return False
|
||||
separator = client_name[-len(canonical) - 1]
|
||||
return separator in ("_", "-")
|
||||
|
||||
def _get_tools_by_names(
|
||||
self, tool_names: List[str], available_tools: List[Any]
|
||||
) -> List[Any]:
|
||||
"""Get tools from available_tools by their names, preserving order."""
|
||||
# Match tools from available_tools (preserves format - dict or MCPTool)
|
||||
matched_tools = []
|
||||
for tool in available_tools:
|
||||
tool_name, _ = self._extract_tool_info(tool)
|
||||
if tool_name in tool_names:
|
||||
matched_tools.append(tool)
|
||||
"""
|
||||
Get tools from available_tools by their names, preserving the
|
||||
semantic router's ordering.
|
||||
|
||||
# Reorder to match semantic router's ordering
|
||||
tool_map = {self._extract_tool_info(t)[0]: t for t in matched_tools}
|
||||
return [tool_map[name] for name in tool_names if name in tool_map]
|
||||
Matching is tolerant of client-side namespace prefixes: if an
|
||||
incoming tool arrived as ``<client_alias>_<canonical>`` while the
|
||||
router returned ``<canonical>`` (see
|
||||
``_name_matches_canonical``), that tool is still selected. The
|
||||
returned tool object is the original from ``available_tools``, so
|
||||
the client-facing name is preserved for tool-call round-trips.
|
||||
"""
|
||||
# Build an index of incoming tools by their client-facing name.
|
||||
# Exact matches win over suffix matches when both are present, and
|
||||
# each incoming tool is returned at most once even if two canonical
|
||||
# names happen to be tail-compatible with the same incoming name.
|
||||
available_by_name: Dict[str, Any] = {}
|
||||
for tool in available_tools:
|
||||
client_name, _ = self._extract_tool_info(tool)
|
||||
if client_name and client_name not in available_by_name:
|
||||
available_by_name[client_name] = tool
|
||||
|
||||
matched: List[Any] = []
|
||||
used_ids: set = set()
|
||||
for canonical in tool_names:
|
||||
tool = available_by_name.get(canonical)
|
||||
if tool is None:
|
||||
# Prefer the shortest qualifying name. When several
|
||||
# incoming tools suffix-match the same canonical (e.g.
|
||||
# "my_search" and "my_tag_search" both end in "search"),
|
||||
# the one closest in length to the canonical is the
|
||||
# least-wrapped and most likely the intended target.
|
||||
best_name: Optional[str] = None
|
||||
for client_name in available_by_name:
|
||||
if not self._name_matches_canonical(client_name, canonical):
|
||||
continue
|
||||
if best_name is None or len(client_name) < len(best_name):
|
||||
best_name = client_name
|
||||
if best_name is not None:
|
||||
tool = available_by_name[best_name]
|
||||
if tool is not None and id(tool) not in used_ids:
|
||||
matched.append(tool)
|
||||
used_ids.add(id(tool))
|
||||
return matched
|
||||
|
||||
def extract_user_query(self, messages: List[Dict[str, Any]]) -> str:
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -7289,6 +7289,7 @@ async def chat_completion( # noqa: PLR0915
|
|||
and user_api_key_dict.agent_id is not None
|
||||
):
|
||||
data["metadata"]["agent_id"] = user_api_key_dict.agent_id
|
||||
|
||||
base_llm_response_processor = ProxyBaseLLMRequestProcessing(data=data)
|
||||
try:
|
||||
result = await base_llm_response_processor.base_process_llm_request(
|
||||
|
|
|
|||
|
|
@ -139,7 +139,9 @@ class ProviderSpecificModelInfo(TypedDict, total=False):
|
|||
supports_reasoning: Optional[bool]
|
||||
supports_url_context: Optional[bool]
|
||||
supports_none_reasoning_effort: Optional[bool]
|
||||
supports_minimal_reasoning_effort: Optional[bool]
|
||||
supports_xhigh_reasoning_effort: Optional[bool]
|
||||
supports_max_reasoning_effort: Optional[bool]
|
||||
|
||||
|
||||
class SearchContextCostPerQuery(TypedDict, total=False):
|
||||
|
|
|
|||
|
|
@ -5893,9 +5893,15 @@ def _get_model_info_helper( # noqa: PLR0915
|
|||
supports_none_reasoning_effort=_model_info.get(
|
||||
"supports_none_reasoning_effort", None
|
||||
),
|
||||
supports_minimal_reasoning_effort=_model_info.get(
|
||||
"supports_minimal_reasoning_effort", None
|
||||
),
|
||||
supports_xhigh_reasoning_effort=_model_info.get(
|
||||
"supports_xhigh_reasoning_effort", None
|
||||
),
|
||||
supports_max_reasoning_effort=_model_info.get(
|
||||
"supports_max_reasoning_effort", None
|
||||
),
|
||||
supports_computer_use=_model_info.get("supports_computer_use", None),
|
||||
search_context_cost_per_query=_model_info.get(
|
||||
"search_context_cost_per_query", None
|
||||
|
|
@ -8946,6 +8952,12 @@ class ProviderConfigManager:
|
|||
)
|
||||
|
||||
return get_openrouter_image_generation_config(model)
|
||||
elif LlmProviders.DASHSCOPE == provider:
|
||||
from litellm.llms.dashscope.image_generation import (
|
||||
get_dashscope_image_generation_config,
|
||||
)
|
||||
|
||||
return get_dashscope_image_generation_config(model)
|
||||
return None
|
||||
|
||||
@staticmethod
|
||||
|
|
|
|||
|
|
@ -1006,7 +1006,8 @@
|
|||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true,
|
||||
"supports_max_reasoning_effort": true
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"global.anthropic.claude-opus-4-6-v1": {
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
|
|
@ -1034,7 +1035,8 @@
|
|||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true,
|
||||
"supports_max_reasoning_effort": true
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"us.anthropic.claude-opus-4-6-v1": {
|
||||
"cache_creation_input_token_cost": 6.875e-06,
|
||||
|
|
@ -1062,7 +1064,8 @@
|
|||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true,
|
||||
"supports_max_reasoning_effort": true
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"eu.anthropic.claude-opus-4-6-v1": {
|
||||
"cache_creation_input_token_cost": 6.875e-06,
|
||||
|
|
@ -1090,7 +1093,8 @@
|
|||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true,
|
||||
"supports_max_reasoning_effort": true
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"au.anthropic.claude-opus-4-6-v1": {
|
||||
"cache_creation_input_token_cost": 6.875e-06,
|
||||
|
|
@ -1118,7 +1122,8 @@
|
|||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true,
|
||||
"supports_max_reasoning_effort": true
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"anthropic.claude-opus-4-7": {
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
|
|
@ -1146,7 +1151,9 @@
|
|||
"supports_vision": true,
|
||||
"supports_xhigh_reasoning_effort": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true
|
||||
"supports_native_structured_output": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"anthropic.claude-mythos-preview": {
|
||||
"input_cost_per_token": 0,
|
||||
|
|
@ -1188,7 +1195,9 @@
|
|||
"supports_vision": true,
|
||||
"supports_xhigh_reasoning_effort": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true
|
||||
"supports_native_structured_output": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"us.anthropic.claude-opus-4-7": {
|
||||
"cache_creation_input_token_cost": 6.875e-06,
|
||||
|
|
@ -1216,7 +1225,9 @@
|
|||
"supports_vision": true,
|
||||
"supports_xhigh_reasoning_effort": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true
|
||||
"supports_native_structured_output": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"eu.anthropic.claude-opus-4-7": {
|
||||
"cache_creation_input_token_cost": 6.875e-06,
|
||||
|
|
@ -1244,7 +1255,9 @@
|
|||
"supports_vision": true,
|
||||
"supports_xhigh_reasoning_effort": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true
|
||||
"supports_native_structured_output": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"au.anthropic.claude-opus-4-7": {
|
||||
"cache_creation_input_token_cost": 6.875e-06,
|
||||
|
|
@ -1272,7 +1285,9 @@
|
|||
"supports_vision": true,
|
||||
"supports_xhigh_reasoning_effort": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true
|
||||
"supports_native_structured_output": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"anthropic.claude-sonnet-4-6": {
|
||||
"cache_creation_input_token_cost": 3.75e-06,
|
||||
|
|
@ -1299,7 +1314,8 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true
|
||||
"supports_native_structured_output": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"global.anthropic.claude-sonnet-4-6": {
|
||||
"cache_creation_input_token_cost": 3.75e-06,
|
||||
|
|
@ -1326,7 +1342,8 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true
|
||||
"supports_native_structured_output": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"us.anthropic.claude-sonnet-4-6": {
|
||||
"cache_creation_input_token_cost": 4.125e-06,
|
||||
|
|
@ -1353,7 +1370,8 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true
|
||||
"supports_native_structured_output": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"eu.anthropic.claude-sonnet-4-6": {
|
||||
"cache_creation_input_token_cost": 4.125e-06,
|
||||
|
|
@ -1380,7 +1398,8 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true
|
||||
"supports_native_structured_output": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"au.anthropic.claude-sonnet-4-6": {
|
||||
"cache_creation_input_token_cost": 4.125e-06,
|
||||
|
|
@ -1407,7 +1426,8 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true
|
||||
"supports_native_structured_output": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"anthropic.claude-sonnet-4-20250514-v1:0": {
|
||||
"cache_creation_input_token_cost": 3.75e-06,
|
||||
|
|
@ -1925,7 +1945,8 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 159,
|
||||
"supports_max_reasoning_effort": true
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"azure_ai/claude-opus-4-7": {
|
||||
"input_cost_per_token": 5e-06,
|
||||
|
|
@ -1953,7 +1974,9 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"supports_xhigh_reasoning_effort": true,
|
||||
"tool_use_system_prompt_tokens": 159
|
||||
"tool_use_system_prompt_tokens": 159,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"azure_ai/claude-opus-4-1": {
|
||||
"cache_creation_input_token_cost": 1.875e-05,
|
||||
|
|
@ -2017,7 +2040,8 @@
|
|||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"azure/computer-use-preview": {
|
||||
"input_cost_per_token": 3e-06,
|
||||
|
|
@ -8923,7 +8947,8 @@
|
|||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"claude-sonnet-4-5-20250929-v1:0": {
|
||||
"cache_creation_input_token_cost": 3.75e-06,
|
||||
|
|
@ -9117,7 +9142,8 @@
|
|||
"us": 1.1,
|
||||
"fast": 6.0
|
||||
},
|
||||
"supports_max_reasoning_effort": true
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"claude-opus-4-6-20260205": {
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
|
|
@ -9149,7 +9175,8 @@
|
|||
"us": 1.1,
|
||||
"fast": 6.0
|
||||
},
|
||||
"supports_max_reasoning_effort": true
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"claude-opus-4-7": {
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
|
|
@ -9181,7 +9208,9 @@
|
|||
"provider_specific_entry": {
|
||||
"us": 1.1,
|
||||
"fast": 6.0
|
||||
}
|
||||
},
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"claude-opus-4-7-20260416": {
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
|
|
@ -9213,7 +9242,9 @@
|
|||
"provider_specific_entry": {
|
||||
"us": 1.1,
|
||||
"fast": 6.0
|
||||
}
|
||||
},
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"claude-sonnet-4-20250514": {
|
||||
"deprecation_date": "2026-05-14",
|
||||
|
|
@ -10366,6 +10397,22 @@
|
|||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"dashscope/qwen-image-2.0": {
|
||||
"litellm_provider": "dashscope",
|
||||
"mode": "image_generation",
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models",
|
||||
"supported_endpoints": [
|
||||
"/v1/images/generations"
|
||||
]
|
||||
},
|
||||
"dashscope/qwen-image-2.0-pro": {
|
||||
"litellm_provider": "dashscope",
|
||||
"mode": "image_generation",
|
||||
"source": "https://www.alibabacloud.com/help/en/model-studio/models",
|
||||
"supported_endpoints": [
|
||||
"/v1/images/generations"
|
||||
]
|
||||
},
|
||||
"databricks/databricks-bge-large-en": {
|
||||
"input_cost_per_token": 1.0003e-07,
|
||||
"input_dbu_cost_per_token": 1.429e-06,
|
||||
|
|
@ -25082,7 +25129,8 @@
|
|||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 159
|
||||
"tool_use_system_prompt_tokens": 159,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"openrouter/anthropic/claude-opus-4.5": {
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
|
|
@ -25120,7 +25168,8 @@
|
|||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"openrouter/anthropic/claude-sonnet-4.5": {
|
||||
"input_cost_per_image": 0.0048,
|
||||
|
|
@ -30170,7 +30219,8 @@
|
|||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true
|
||||
"supports_vision": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"vercel_ai_gateway/anthropic/claude-sonnet-4": {
|
||||
"cache_creation_input_token_cost": 3.75e-06,
|
||||
|
|
@ -31397,7 +31447,8 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_max_reasoning_effort": true
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"vertex_ai/claude-opus-4-6@default": {
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
|
|
@ -31424,7 +31475,8 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_max_reasoning_effort": true
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"vertex_ai/claude-opus-4-7": {
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
|
|
@ -31451,7 +31503,9 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"supports_xhigh_reasoning_effort": true,
|
||||
"tool_use_system_prompt_tokens": 346
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"vertex_ai/claude-opus-4-7@default": {
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
|
|
@ -31478,7 +31532,9 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"supports_xhigh_reasoning_effort": true,
|
||||
"tool_use_system_prompt_tokens": 346
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"vertex_ai/claude-sonnet-4-5": {
|
||||
"cache_creation_input_token_cost": 3.75e-06,
|
||||
|
|
@ -31530,7 +31586,8 @@
|
|||
"search_context_size_high": 0.01,
|
||||
"search_context_size_low": 0.01,
|
||||
"search_context_size_medium": 0.01
|
||||
}
|
||||
},
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"vertex_ai/claude-sonnet-4-5@20250929": {
|
||||
"cache_creation_input_token_cost": 3.75e-06,
|
||||
|
|
@ -38424,7 +38481,8 @@
|
|||
"search_context_size_high": 0.01,
|
||||
"search_context_size_low": 0.01,
|
||||
"search_context_size_medium": 0.01
|
||||
}
|
||||
},
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"duckduckgo/search": {
|
||||
"litellm_provider": "duckduckgo",
|
||||
|
|
|
|||
|
|
@ -11,6 +11,7 @@ sys.path.insert(
|
|||
|
||||
from litellm.litellm_core_utils.prompt_templates.common_utils import (
|
||||
add_system_prompt_to_messages,
|
||||
get_file_ids_from_messages,
|
||||
get_format_from_file_id,
|
||||
handle_any_messages_to_chat_completion_str_messages_conversion,
|
||||
split_concatenated_json_objects,
|
||||
|
|
@ -254,3 +255,115 @@ def test_split_concatenated_json_invalid_raises():
|
|||
"""Completely invalid JSON raises JSONDecodeError."""
|
||||
with pytest.raises(json.JSONDecodeError):
|
||||
split_concatenated_json_objects("not json at all")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Regression tests for non-OpenAI file content blocks.
|
||||
#
|
||||
# `type: "file"` is a public content-block discriminator. Several producers
|
||||
# (LangChain v1, provider-native shapes, custom user code) emit blocks with
|
||||
# `type: "file"` but without the OpenAI Chat Completions `file` sub-dict.
|
||||
# The discovery helpers below are used unconditionally inside
|
||||
# `AnthropicConfig.validate_environment`, so any crash there surfaces as a
|
||||
# `500 InternalServerError` before the request is even dispatched.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_get_file_ids_from_messages_skips_langchain_v1_file_block():
|
||||
"""A LangChain v1 standardized file block must not crash file-id discovery."""
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "text", "text": "summarise this PDF"},
|
||||
# LangChain v1 shape produced by `_normalize_messages`.
|
||||
# No `file` sub-dict: the discriminator is `type: "file"` but
|
||||
# the payload lives on `base64`/`mime_type` siblings.
|
||||
{
|
||||
"type": "file",
|
||||
"id": "lc_1",
|
||||
"base64": "JVBERi0xLjQK",
|
||||
"mime_type": "application/pdf",
|
||||
"extras": {"file_format": "application/pdf"},
|
||||
},
|
||||
],
|
||||
}
|
||||
]
|
||||
|
||||
assert get_file_ids_from_messages(messages) == []
|
||||
|
||||
|
||||
def test_get_file_ids_from_messages_still_extracts_from_openai_shape():
|
||||
"""Well-formed OpenAI file blocks still yield their file_id."""
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "text", "text": "what is this?"},
|
||||
{"type": "file", "file": {"file_id": "file-abc"}},
|
||||
],
|
||||
}
|
||||
]
|
||||
|
||||
assert get_file_ids_from_messages(messages) == ["file-abc"]
|
||||
|
||||
|
||||
def test_get_file_ids_from_messages_mixed_shapes():
|
||||
"""Mixed OpenAI and non-OpenAI file blocks: extract from the former,
|
||||
ignore the latter."""
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "file", "file": {"file_id": "file-keep"}},
|
||||
{
|
||||
"type": "file",
|
||||
"id": "lc_2",
|
||||
"base64": "AAA",
|
||||
"mime_type": "application/pdf",
|
||||
},
|
||||
],
|
||||
}
|
||||
]
|
||||
|
||||
assert get_file_ids_from_messages(messages) == ["file-keep"]
|
||||
|
||||
|
||||
def test_get_file_ids_from_messages_file_field_not_dict():
|
||||
"""`file` set to a non-dict value (e.g. stringified payload) must not crash."""
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "file", "file": "unexpectedly-a-string"},
|
||||
],
|
||||
}
|
||||
]
|
||||
|
||||
assert get_file_ids_from_messages(messages) == []
|
||||
|
||||
|
||||
def test_update_messages_with_model_file_ids_skips_non_openai_file_blocks():
|
||||
"""`update_messages_with_model_file_ids` is also called on user content
|
||||
before provider dispatch. It must tolerate non-OpenAI file blocks the same
|
||||
way."""
|
||||
langchain_v1_block = {
|
||||
"type": "file",
|
||||
"id": "lc_3",
|
||||
"base64": "AAA",
|
||||
"mime_type": "application/pdf",
|
||||
}
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "text", "text": "hello"},
|
||||
langchain_v1_block,
|
||||
],
|
||||
}
|
||||
]
|
||||
|
||||
updated = update_messages_with_model_file_ids(messages, "model-1", {})
|
||||
|
||||
# Messages pass through unchanged when there is no `file` sub-dict to remap.
|
||||
assert updated == messages
|
||||
|
|
|
|||
|
|
@ -2162,16 +2162,19 @@ class TestTranslateAnthropicOutputFormatToOpenAI:
|
|||
assert sorted(schema["required"]) == ["age", "email", "name"]
|
||||
|
||||
def test_invalid_output_format_returns_none(self):
|
||||
assert (
|
||||
self.adapter.translate_anthropic_output_format_to_openai("invalid") is None
|
||||
)
|
||||
assert (
|
||||
self.adapter.translate_anthropic_output_format_to_openai({"type": "text"})
|
||||
is None
|
||||
)
|
||||
assert (
|
||||
self.adapter.translate_anthropic_output_format_to_openai(
|
||||
{"type": "json_schema"}
|
||||
)
|
||||
is None
|
||||
)
|
||||
assert self.adapter.translate_anthropic_output_format_to_openai("invalid") is None
|
||||
assert self.adapter.translate_anthropic_output_format_to_openai({"type": "text"}) is None
|
||||
assert self.adapter.translate_anthropic_output_format_to_openai({"type": "json_schema"}) is None
|
||||
|
||||
|
||||
def test_translate_anthropic_tool_choice_none():
|
||||
"""
|
||||
Regression test for issue #24443.
|
||||
|
||||
tool_choice={"type": "none"} should be translated to "none" for OpenAI format,
|
||||
not raise a ValueError.
|
||||
"""
|
||||
adapter = LiteLLMAnthropicMessagesAdapter()
|
||||
|
||||
result = adapter.translate_anthropic_tool_choice_to_openai({"type": "none"})
|
||||
assert result == "none"
|
||||
|
|
|
|||
|
|
@ -0,0 +1,173 @@
|
|||
"""
|
||||
Tests for reasoning_auto_summary support on the native /v1/messages handler.
|
||||
|
||||
When reasoning_auto_summary is enabled (via litellm.reasoning_auto_summary or
|
||||
LITELLM_REASONING_AUTO_SUMMARY env var), the handler injects
|
||||
thinking.display = "summarized" into the request params for active thinking
|
||||
modes (type="enabled" or type="adaptive").
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
sys.path.insert(0, os.path.abspath("../../../../.."))
|
||||
|
||||
import litellm
|
||||
from litellm.llms.anthropic.experimental_pass_through.messages.handler import (
|
||||
anthropic_messages_handler,
|
||||
)
|
||||
|
||||
|
||||
def _call_handler_and_capture_optional_params(thinking=None, **extra_kwargs):
|
||||
"""
|
||||
Call anthropic_messages_handler with an Anthropic model and capture the
|
||||
anthropic_messages_optional_request_params dict passed to
|
||||
base_llm_http_handler.anthropic_messages_handler.
|
||||
|
||||
Returns the captured dict.
|
||||
"""
|
||||
captured = {}
|
||||
|
||||
with patch(
|
||||
"litellm.llms.anthropic.experimental_pass_through.messages.handler."
|
||||
"base_llm_http_handler"
|
||||
) as mock_handler, patch(
|
||||
"litellm.llms.anthropic.experimental_pass_through.messages.handler."
|
||||
"ProviderConfigManager"
|
||||
) as mock_pcm:
|
||||
# Make get_provider_anthropic_messages_config return a non-None config
|
||||
# so the handler takes the native Anthropic path
|
||||
mock_pcm.get_provider_anthropic_messages_config.return_value = MagicMock()
|
||||
mock_handler.anthropic_messages_handler.return_value = MagicMock()
|
||||
|
||||
kwargs = dict(extra_kwargs)
|
||||
if thinking is not None:
|
||||
kwargs["thinking"] = thinking
|
||||
|
||||
try:
|
||||
anthropic_messages_handler(
|
||||
max_tokens=1024,
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
model="claude-sonnet-4-20250514",
|
||||
custom_llm_provider="anthropic",
|
||||
api_key="test-key",
|
||||
**kwargs,
|
||||
)
|
||||
except (ValueError, TypeError, AttributeError):
|
||||
pass
|
||||
|
||||
if mock_handler.anthropic_messages_handler.called:
|
||||
captured = mock_handler.anthropic_messages_handler.call_args.kwargs.get(
|
||||
"anthropic_messages_optional_request_params", {}
|
||||
)
|
||||
|
||||
return captured
|
||||
|
||||
|
||||
class TestReasoningAutoSummaryMessages:
|
||||
"""Tests for thinking.display injection on native /v1/messages handler."""
|
||||
|
||||
def test_adaptive_thinking_gets_display_summarized(self):
|
||||
"""reasoning_auto_summary=True + thinking.type='adaptive' -> display='summarized'."""
|
||||
with patch.object(litellm, "reasoning_auto_summary", True):
|
||||
params = _call_handler_and_capture_optional_params(
|
||||
thinking={"type": "adaptive", "budget_tokens": 5000}
|
||||
)
|
||||
thinking = params.get("thinking", {})
|
||||
assert thinking.get("display") == "summarized"
|
||||
assert thinking.get("type") == "adaptive"
|
||||
assert thinking.get("budget_tokens") == 5000
|
||||
|
||||
def test_enabled_thinking_gets_display_summarized(self):
|
||||
"""reasoning_auto_summary=True + thinking.type='enabled' -> display='summarized'."""
|
||||
with patch.object(litellm, "reasoning_auto_summary", True):
|
||||
params = _call_handler_and_capture_optional_params(
|
||||
thinking={"type": "enabled", "budget_tokens": 10000}
|
||||
)
|
||||
thinking = params.get("thinking", {})
|
||||
assert thinking.get("display") == "summarized"
|
||||
assert thinking.get("type") == "enabled"
|
||||
|
||||
def test_disabled_thinking_no_display(self):
|
||||
"""reasoning_auto_summary=True + thinking.type='disabled' -> display NOT set."""
|
||||
with patch.object(litellm, "reasoning_auto_summary", True):
|
||||
params = _call_handler_and_capture_optional_params(
|
||||
thinking={"type": "disabled"}
|
||||
)
|
||||
thinking = params.get("thinking", {})
|
||||
assert "display" not in thinking
|
||||
|
||||
def test_no_injection_when_flag_false(self):
|
||||
"""reasoning_auto_summary=False + active thinking -> display NOT set."""
|
||||
with patch.object(litellm, "reasoning_auto_summary", False):
|
||||
params = _call_handler_and_capture_optional_params(
|
||||
thinking={"type": "enabled", "budget_tokens": 10000}
|
||||
)
|
||||
thinking = params.get("thinking", {})
|
||||
assert "display" not in thinking
|
||||
|
||||
def test_no_thinking_param_no_crash(self):
|
||||
"""reasoning_auto_summary=True but no thinking param -> nothing changes."""
|
||||
with patch.object(litellm, "reasoning_auto_summary", True):
|
||||
params = _call_handler_and_capture_optional_params()
|
||||
thinking = params.get("thinking")
|
||||
if thinking is not None:
|
||||
assert "display" not in thinking
|
||||
|
||||
def test_env_var_enables_auto_summary(self):
|
||||
"""LITELLM_REASONING_AUTO_SUMMARY=true env var enables the feature."""
|
||||
with patch.object(litellm, "reasoning_auto_summary", False), patch.dict(
|
||||
os.environ, {"LITELLM_REASONING_AUTO_SUMMARY": "true"}
|
||||
):
|
||||
params = _call_handler_and_capture_optional_params(
|
||||
thinking={"type": "adaptive", "budget_tokens": 5000}
|
||||
)
|
||||
thinking = params.get("thinking", {})
|
||||
assert thinking.get("display") == "summarized"
|
||||
|
||||
def test_existing_display_summarized_preserved(self):
|
||||
"""User already passes display='summarized' -> preserved as-is."""
|
||||
with patch.object(litellm, "reasoning_auto_summary", True):
|
||||
params = _call_handler_and_capture_optional_params(
|
||||
thinking={
|
||||
"type": "enabled",
|
||||
"budget_tokens": 10000,
|
||||
"display": "summarized",
|
||||
}
|
||||
)
|
||||
thinking = params.get("thinking", {})
|
||||
assert thinking.get("display") == "summarized"
|
||||
|
||||
def test_existing_display_summarized_without_flag(self):
|
||||
"""User passes display='summarized' + flag=False -> preserved as-is."""
|
||||
with patch.object(litellm, "reasoning_auto_summary", False):
|
||||
params = _call_handler_and_capture_optional_params(
|
||||
thinking={
|
||||
"type": "enabled",
|
||||
"budget_tokens": 10000,
|
||||
"display": "summarized",
|
||||
}
|
||||
)
|
||||
thinking = params.get("thinking", {})
|
||||
assert thinking.get("display") == "summarized"
|
||||
|
||||
def test_omitted_overridden_to_summarized(self):
|
||||
"""User passes display='omitted' + reasoning_auto_summary=True -> overridden.
|
||||
|
||||
Documents current behavior: the code unconditionally sets
|
||||
display='summarized' when auto_summary is enabled and thinking is active,
|
||||
regardless of any pre-existing display value.
|
||||
"""
|
||||
with patch.object(litellm, "reasoning_auto_summary", True):
|
||||
params = _call_handler_and_capture_optional_params(
|
||||
thinking={
|
||||
"type": "enabled",
|
||||
"budget_tokens": 10000,
|
||||
"display": "omitted",
|
||||
}
|
||||
)
|
||||
thinking = params.get("thinking", {})
|
||||
assert thinking.get("display") == "summarized"
|
||||
|
|
@ -0,0 +1,287 @@
|
|||
"""
|
||||
Tests for reasoning effort capability fields and normalize_reasoning_effort_value.
|
||||
|
||||
Covers:
|
||||
- Commit 1: get_model_info returns supports_minimal/supports_max fields
|
||||
- Commit 2: Model registry entries have correct reasoning effort fields
|
||||
- Commit 3: normalize_reasoning_effort_value degradation chains + adapter translation
|
||||
"""
|
||||
|
||||
import json
|
||||
import os
|
||||
from typing import Any, Dict, Optional
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
|
||||
from litellm.llms.anthropic.experimental_pass_through.utils import (
|
||||
normalize_reasoning_effort_value,
|
||||
)
|
||||
from litellm.utils import get_model_info
|
||||
|
||||
|
||||
def _load_model_registry() -> Dict[str, Any]:
|
||||
"""Load the root model_prices_and_context_window.json."""
|
||||
json_path = os.path.join(
|
||||
os.path.dirname(__file__),
|
||||
"../../../../../model_prices_and_context_window.json",
|
||||
)
|
||||
with open(json_path) as f:
|
||||
return json.load(f)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Commit 1: get_model_info returns supports_minimal and supports_max fields
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestGetModelInfoReasoningEffortFields:
|
||||
"""get_model_info should expose supports_minimal_reasoning_effort and
|
||||
supports_max_reasoning_effort from the model registry."""
|
||||
|
||||
def test_opus_4_6_has_supports_minimal(self):
|
||||
info = get_model_info("claude-opus-4-6")
|
||||
assert "supports_minimal_reasoning_effort" in info
|
||||
|
||||
def test_opus_4_6_has_supports_max(self):
|
||||
info = get_model_info("claude-opus-4-6")
|
||||
assert "supports_max_reasoning_effort" in info
|
||||
|
||||
def test_opus_4_7_has_supports_minimal(self):
|
||||
info = get_model_info("claude-opus-4-7")
|
||||
assert "supports_minimal_reasoning_effort" in info
|
||||
|
||||
def test_opus_4_7_has_supports_max(self):
|
||||
info = get_model_info("claude-opus-4-7")
|
||||
assert "supports_max_reasoning_effort" in info
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Commit 2: JSON registry has correct reasoning effort fields
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestModelRegistryReasoningEffortFields:
|
||||
"""Verify specific models have the expected reasoning effort capability
|
||||
values in the JSON registry file."""
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _load_registry(self):
|
||||
self.registry = _load_model_registry()
|
||||
|
||||
def test_opus_4_7_supports_max(self):
|
||||
entry = self.registry["claude-opus-4-7"]
|
||||
assert entry.get("supports_max_reasoning_effort") is True
|
||||
|
||||
def test_opus_4_6_supports_max(self):
|
||||
entry = self.registry["claude-opus-4-6"]
|
||||
assert entry.get("supports_max_reasoning_effort") is True
|
||||
|
||||
def test_opus_4_7_supports_minimal(self):
|
||||
entry = self.registry["claude-opus-4-7"]
|
||||
assert entry.get("supports_minimal_reasoning_effort") is True
|
||||
|
||||
def test_opus_4_6_supports_minimal(self):
|
||||
entry = self.registry["claude-opus-4-6"]
|
||||
assert entry.get("supports_minimal_reasoning_effort") is True
|
||||
|
||||
def test_sonnet_4_6_supports_minimal(self):
|
||||
entry = self.registry["anthropic.claude-sonnet-4-6"]
|
||||
assert entry.get("supports_minimal_reasoning_effort") is True
|
||||
|
||||
def test_bedrock_opus_4_7_supports_max(self):
|
||||
entry = self.registry["anthropic.claude-opus-4-7"]
|
||||
assert entry.get("supports_max_reasoning_effort") is True
|
||||
assert entry.get("supports_minimal_reasoning_effort") is True
|
||||
|
||||
def test_vertex_opus_4_7_supports_max(self):
|
||||
entry = self.registry["vertex_ai/claude-opus-4-7"]
|
||||
assert entry.get("supports_max_reasoning_effort") is True
|
||||
assert entry.get("supports_minimal_reasoning_effort") is True
|
||||
|
||||
def test_vertex_opus_4_6_supports_max(self):
|
||||
entry = self.registry["vertex_ai/claude-opus-4-6"]
|
||||
assert entry.get("supports_max_reasoning_effort") is True
|
||||
assert entry.get("supports_minimal_reasoning_effort") is True
|
||||
|
||||
def test_azure_ai_opus_4_6_supports_minimal(self):
|
||||
entry = self.registry["azure_ai/claude-opus-4-6"]
|
||||
assert entry.get("supports_minimal_reasoning_effort") is True
|
||||
|
||||
def test_azure_ai_opus_4_7_supports_max(self):
|
||||
entry = self.registry["azure_ai/claude-opus-4-7"]
|
||||
assert entry.get("supports_max_reasoning_effort") is True
|
||||
assert entry.get("supports_minimal_reasoning_effort") is True
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Commit 3: normalize_reasoning_effort_value
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _mock_model_info(**flags):
|
||||
"""Return a mock model_info dict with given capability flags."""
|
||||
return flags
|
||||
|
||||
|
||||
class TestNormalizeReasoningEffortValue:
|
||||
"""Test degradation chains for normalize_reasoning_effort_value."""
|
||||
|
||||
# --- "max" degradation chain ---
|
||||
|
||||
def test_max_stays_max_when_supported(self):
|
||||
with patch(
|
||||
"litellm.utils.get_model_info",
|
||||
return_value=_mock_model_info(
|
||||
supports_max_reasoning_effort=True,
|
||||
supports_xhigh_reasoning_effort=True,
|
||||
),
|
||||
):
|
||||
assert normalize_reasoning_effort_value("max", model="test") == "max"
|
||||
|
||||
def test_max_degrades_to_xhigh(self):
|
||||
with patch(
|
||||
"litellm.utils.get_model_info",
|
||||
return_value=_mock_model_info(
|
||||
supports_max_reasoning_effort=False,
|
||||
supports_xhigh_reasoning_effort=True,
|
||||
),
|
||||
):
|
||||
assert normalize_reasoning_effort_value("max", model="test") == "xhigh"
|
||||
|
||||
def test_max_degrades_to_high(self):
|
||||
with patch(
|
||||
"litellm.utils.get_model_info",
|
||||
return_value=_mock_model_info(
|
||||
supports_max_reasoning_effort=False,
|
||||
supports_xhigh_reasoning_effort=False,
|
||||
),
|
||||
):
|
||||
assert normalize_reasoning_effort_value("max", model="test") == "high"
|
||||
|
||||
# --- "xhigh" degradation chain ---
|
||||
|
||||
def test_xhigh_stays_xhigh_when_supported(self):
|
||||
with patch(
|
||||
"litellm.utils.get_model_info",
|
||||
return_value=_mock_model_info(supports_xhigh_reasoning_effort=True),
|
||||
):
|
||||
assert normalize_reasoning_effort_value("xhigh", model="test") == "xhigh"
|
||||
|
||||
def test_xhigh_degrades_to_high(self):
|
||||
with patch(
|
||||
"litellm.utils.get_model_info",
|
||||
return_value=_mock_model_info(supports_xhigh_reasoning_effort=False),
|
||||
):
|
||||
assert normalize_reasoning_effort_value("xhigh", model="test") == "high"
|
||||
|
||||
# --- "minimal" degradation chain ---
|
||||
|
||||
def test_minimal_stays_minimal_when_supported(self):
|
||||
with patch(
|
||||
"litellm.utils.get_model_info",
|
||||
return_value=_mock_model_info(supports_minimal_reasoning_effort=True),
|
||||
):
|
||||
assert (
|
||||
normalize_reasoning_effort_value("minimal", model="test") == "minimal"
|
||||
)
|
||||
|
||||
def test_minimal_degrades_to_low(self):
|
||||
with patch(
|
||||
"litellm.utils.get_model_info",
|
||||
return_value=_mock_model_info(supports_minimal_reasoning_effort=False),
|
||||
):
|
||||
assert normalize_reasoning_effort_value("minimal", model="test") == "low"
|
||||
|
||||
# --- passthrough values ---
|
||||
|
||||
def test_high_passes_through(self):
|
||||
assert normalize_reasoning_effort_value("high", model="test") == "high"
|
||||
|
||||
def test_medium_passes_through(self):
|
||||
assert normalize_reasoning_effort_value("medium", model="test") == "medium"
|
||||
|
||||
def test_low_passes_through(self):
|
||||
assert normalize_reasoning_effort_value("low", model="test") == "low"
|
||||
|
||||
# --- exception fallback ---
|
||||
|
||||
def test_exception_fallback_uses_empty_model_info(self):
|
||||
"""When get_model_info raises, treat model_info as {} (no capabilities)."""
|
||||
with patch(
|
||||
"litellm.utils.get_model_info",
|
||||
side_effect=Exception("model not found"),
|
||||
):
|
||||
# "max" with no capabilities -> "high"
|
||||
assert normalize_reasoning_effort_value("max", model="unknown") == "high"
|
||||
# "minimal" with no capabilities -> "low"
|
||||
assert normalize_reasoning_effort_value("minimal", model="unknown") == "low"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Commit 3: Adapter translation — adaptive thinking + output_config.effort
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestAdapterAdaptiveThinking:
|
||||
"""Test that adaptive thinking type maps correctly through the adapters."""
|
||||
|
||||
def test_messages_adapter_adaptive_returns_medium_default(self):
|
||||
"""Adaptive thinking returns 'medium' as default reasoning_effort."""
|
||||
from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import (
|
||||
LiteLLMAnthropicMessagesAdapter,
|
||||
)
|
||||
|
||||
adapter = LiteLLMAnthropicMessagesAdapter()
|
||||
result = adapter.translate_anthropic_thinking_to_reasoning_effort(
|
||||
{"type": "adaptive"}
|
||||
)
|
||||
assert result == "medium"
|
||||
|
||||
def test_messages_adapter_adaptive_overridden_by_output_config(self):
|
||||
"""For adaptive thinking, output_config.effort overrides reasoning_effort."""
|
||||
from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import (
|
||||
LiteLLMAnthropicMessagesAdapter,
|
||||
)
|
||||
from litellm.types.llms.anthropic import AnthropicMessagesRequest
|
||||
|
||||
adapter = LiteLLMAnthropicMessagesAdapter()
|
||||
request = AnthropicMessagesRequest(
|
||||
model="test-model",
|
||||
messages=[{"role": "user", "content": "hello"}],
|
||||
max_tokens=1024,
|
||||
thinking={"type": "adaptive"},
|
||||
output_config={"effort": "high"},
|
||||
)
|
||||
openai_kwargs, _ = adapter.translate_anthropic_to_openai(request)
|
||||
# reasoning_effort should be set (either as string or dict with effort)
|
||||
re = openai_kwargs.get("reasoning_effort")
|
||||
if isinstance(re, dict):
|
||||
assert re["effort"] == "high"
|
||||
else:
|
||||
assert re == "high"
|
||||
|
||||
def test_responses_adapter_adaptive_with_output_config(self):
|
||||
"""Responses adapter: adaptive thinking + output_config.effort."""
|
||||
from litellm.llms.anthropic.experimental_pass_through.responses_adapters.transformation import (
|
||||
LiteLLMAnthropicToResponsesAPIAdapter,
|
||||
)
|
||||
|
||||
result = LiteLLMAnthropicToResponsesAPIAdapter.translate_thinking_to_reasoning(
|
||||
thinking={"type": "adaptive"},
|
||||
output_config={"effort": "xhigh"},
|
||||
)
|
||||
assert result is not None
|
||||
assert result["effort"] == "xhigh"
|
||||
|
||||
def test_responses_adapter_adaptive_default_medium(self):
|
||||
"""Responses adapter: adaptive thinking without output_config defaults to medium."""
|
||||
from litellm.llms.anthropic.experimental_pass_through.responses_adapters.transformation import (
|
||||
LiteLLMAnthropicToResponsesAPIAdapter,
|
||||
)
|
||||
|
||||
result = LiteLLMAnthropicToResponsesAPIAdapter.translate_thinking_to_reasoning(
|
||||
thinking={"type": "adaptive"},
|
||||
)
|
||||
assert result is not None
|
||||
assert result["effort"] == "medium"
|
||||
151
tests/test_litellm/llms/openai/test_is_model_gpt_5_model.py
Normal file
151
tests/test_litellm/llms/openai/test_is_model_gpt_5_model.py
Normal file
|
|
@ -0,0 +1,151 @@
|
|||
"""
|
||||
Regression tests for is_model_gpt_5_model() in both OpenAI and Azure GPT-5 config
|
||||
classes.
|
||||
|
||||
Background
|
||||
----------
|
||||
In v1.82.3 a substring check was introduced::
|
||||
|
||||
return "gpt-5" in model and "gpt-5-chat" not in model
|
||||
|
||||
This inadvertently treated versioned chat models like ``gpt-5.3-chat`` and
|
||||
``gpt-5.1-chat`` as *non*-GPT-5 models, because the string ``"gpt-5-chat"`` is
|
||||
a substring of ``"gpt-5.3-chat"``. Those models were then routed through the
|
||||
regular Azure chat path which does not suppress ``parallel_tool_calls``, causing
|
||||
Azure to return ``finish_reason="stop"`` together with tool_calls and breaking
|
||||
n8n AI-agent workflows.
|
||||
|
||||
There are two distinct families:
|
||||
|
||||
* **gpt-5-chat family** (``gpt-5-chat``, ``gpt-5-chat-latest``,
|
||||
``gpt-5-chat-2025-08-07``, …) — regular chat models that support ``temperature``
|
||||
and ``tool_choice`` but NOT ``reasoning_effort``. Must NOT be on the GPT-5
|
||||
reasoning path.
|
||||
|
||||
* **Versioned chat models** (``gpt-5.1-chat``, ``gpt-5.2-chat``,
|
||||
``gpt-5.3-chat``, …) — ARE GPT-5 reasoning models and must stay on the GPT-5
|
||||
path.
|
||||
|
||||
The fix uses a prefix check (``startswith("gpt-5-chat")``) on the normalised model
|
||||
name instead of a substring check, which correctly distinguishes the two families.
|
||||
"""
|
||||
|
||||
import pytest
|
||||
|
||||
from litellm.llms.openai.chat.gpt_5_transformation import OpenAIGPT5Config
|
||||
from litellm.llms.azure.chat.gpt_5_transformation import AzureOpenAIGPT5Config
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Parametrized fixtures
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
# Models that MUST be classified as GPT-5 (routed through GPT-5 reasoning path)
|
||||
GPT5_MODELS = [
|
||||
"gpt-5",
|
||||
"gpt-5.1",
|
||||
"gpt-5.2",
|
||||
"gpt-5.3",
|
||||
"gpt-5.4",
|
||||
"gpt-5.1-chat", # versioned chat — THE KEY REGRESSION CASE
|
||||
"gpt-5.2-chat", # versioned chat — also a regression case
|
||||
"gpt-5.3-chat", # versioned chat — THE KEY REGRESSION CASE
|
||||
"gpt-5.2-chat-latest", # versioned chat with date suffix
|
||||
"gpt-5.1-codex",
|
||||
"gpt-5.1-codex-mini",
|
||||
"gpt-5.1-mini",
|
||||
"gpt-5-nano",
|
||||
"gpt-5-mini",
|
||||
"gpt-5-codex",
|
||||
]
|
||||
|
||||
# Models that must NOT be classified as GPT-5 (regular chat path)
|
||||
NON_GPT5_MODELS = [
|
||||
"gpt-5-chat", # gpt-5-chat family — regular chat path
|
||||
"gpt-5-chat-latest", # gpt-5-chat family with alias suffix
|
||||
"gpt-5-chat-2025-08-07", # gpt-5-chat family with date suffix
|
||||
"gpt-4",
|
||||
"gpt-4o",
|
||||
"gpt-4-turbo",
|
||||
"gpt-3.5-turbo",
|
||||
"o1",
|
||||
"o3",
|
||||
"o3-mini",
|
||||
]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# OpenAIGPT5Config
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestOpenAIGPT5ConfigIsModelGpt5Model:
|
||||
|
||||
@pytest.mark.parametrize("model", GPT5_MODELS)
|
||||
def test_gpt5_models_are_classified_as_gpt5(self, model: str):
|
||||
assert OpenAIGPT5Config.is_model_gpt_5_model(
|
||||
model
|
||||
), f"Expected '{model}' to be classified as a GPT-5 model"
|
||||
|
||||
@pytest.mark.parametrize("model", NON_GPT5_MODELS)
|
||||
def test_non_gpt5_models_are_not_classified_as_gpt5(self, model: str):
|
||||
assert not OpenAIGPT5Config.is_model_gpt_5_model(
|
||||
model
|
||||
), f"Expected '{model}' NOT to be classified as a GPT-5 model"
|
||||
|
||||
def test_versioned_chat_models_are_not_excluded_by_prefix(self):
|
||||
"""Core regression guard: gpt-5-chat prefix must not match versioned models."""
|
||||
versioned_chat_models = ["gpt-5.1-chat", "gpt-5.2-chat", "gpt-5.3-chat"]
|
||||
for model in versioned_chat_models:
|
||||
assert OpenAIGPT5Config.is_model_gpt_5_model(
|
||||
model
|
||||
), f"Regression: '{model}' was incorrectly excluded from GPT-5 path"
|
||||
|
||||
def test_gpt5_chat_family_is_excluded(self):
|
||||
"""gpt-5-chat family should stay on the regular chat path."""
|
||||
for model in ["gpt-5-chat", "gpt-5-chat-latest", "gpt-5-chat-2025-08-07"]:
|
||||
assert not OpenAIGPT5Config.is_model_gpt_5_model(
|
||||
model
|
||||
), f"Expected '{model}' (gpt-5-chat family) NOT to be on the GPT-5 path"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# AzureOpenAIGPT5Config
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestAzureOpenAIGPT5ConfigIsModelGpt5Model:
|
||||
|
||||
@pytest.mark.parametrize("model", GPT5_MODELS)
|
||||
def test_gpt5_models_are_classified_as_gpt5(self, model: str):
|
||||
assert AzureOpenAIGPT5Config.is_model_gpt_5_model(
|
||||
model
|
||||
), f"Expected Azure '{model}' to be classified as a GPT-5 model"
|
||||
|
||||
@pytest.mark.parametrize("model", NON_GPT5_MODELS)
|
||||
def test_non_gpt5_models_are_not_classified_as_gpt5(self, model: str):
|
||||
assert not AzureOpenAIGPT5Config.is_model_gpt_5_model(
|
||||
model
|
||||
), f"Expected Azure '{model}' NOT to be classified as a GPT-5 model"
|
||||
|
||||
def test_versioned_chat_models_are_not_excluded_by_prefix(self):
|
||||
"""Core regression guard: gpt-5-chat prefix must not match versioned models."""
|
||||
versioned_chat_models = ["gpt-5.1-chat", "gpt-5.2-chat", "gpt-5.3-chat"]
|
||||
for model in versioned_chat_models:
|
||||
assert AzureOpenAIGPT5Config.is_model_gpt_5_model(
|
||||
model
|
||||
), f"Regression: Azure '{model}' was incorrectly excluded from GPT-5 path"
|
||||
|
||||
def test_gpt5_chat_family_is_excluded(self):
|
||||
"""gpt-5-chat family should stay on the regular chat path."""
|
||||
for model in ["gpt-5-chat", "gpt-5-chat-latest", "gpt-5-chat-2025-08-07"]:
|
||||
assert not AzureOpenAIGPT5Config.is_model_gpt_5_model(
|
||||
model
|
||||
), f"Expected Azure '{model}' (gpt-5-chat family) NOT to be on the GPT-5 path"
|
||||
|
||||
def test_gpt5_series_routing_prefix_is_always_classified_as_gpt5(self):
|
||||
"""Models using the gpt5_series/ manual-routing prefix must always match."""
|
||||
series_models = ["gpt5_series/my-deployment", "gpt5_series/prod"]
|
||||
for model in series_models:
|
||||
assert AzureOpenAIGPT5Config.is_model_gpt_5_model(
|
||||
model
|
||||
), f"Azure '{model}' with gpt5_series/ prefix should be classified as GPT-5"
|
||||
|
|
@ -8,6 +8,7 @@ import sys
|
|||
import pytest
|
||||
|
||||
from litellm.llms.ovhcloud.utils import OVHCloudException
|
||||
from litellm.utils import get_optional_params
|
||||
|
||||
sys.path.insert(
|
||||
0, os.path.abspath("../../../../..")
|
||||
|
|
@ -144,6 +145,38 @@ class TestOVHCloudConfig:
|
|||
assert error.message == "Test error"
|
||||
assert error.status_code == 400
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model",
|
||||
[
|
||||
"Meta-Llama-3_3-70B-Instruct",
|
||||
"Meta-Llama-3_1-70B-Instruct",
|
||||
"Mixtral-8x7B-Instruct-v0.1",
|
||||
"gpt-oss-120b",
|
||||
"some-model-not-in-the-cost-map",
|
||||
],
|
||||
)
|
||||
def test_tools_not_filtered_by_static_model_map(self, model):
|
||||
"""
|
||||
OVHCloud AI Endpoints are OpenAI-compatible; tools/tool_choice must pass
|
||||
through for any model. The server is responsible for rejecting unsupported
|
||||
tool calls — LiteLLM must not strip them based on a stale static catalog.
|
||||
"""
|
||||
|
||||
params = get_optional_params(
|
||||
model=model,
|
||||
custom_llm_provider="ovhcloud",
|
||||
tools=[
|
||||
{
|
||||
"type": "function",
|
||||
"function": {"name": "x", "parameters": {}},
|
||||
}
|
||||
],
|
||||
tool_choice="auto",
|
||||
)
|
||||
|
||||
assert "tools" in params
|
||||
assert "tool_choice" in params
|
||||
|
||||
|
||||
def test_ovhcloud_integration():
|
||||
import os
|
||||
|
|
|
|||
|
|
@ -967,6 +967,94 @@ class TestMediaResolution:
|
|||
assert "mediaResolution" not in result["generationConfig"]
|
||||
|
||||
|
||||
# Tests for VideoMetadata support across all Gemini models (Issue #25474)
|
||||
class TestVideoMetadataAllGeminiModels:
|
||||
"""Tests that video_metadata (fps, start_offset, end_offset) works for all Gemini models"""
|
||||
|
||||
def _make_video_messages(self, video_metadata: dict) -> list:
|
||||
return [
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "text", "text": "Analyze this video"},
|
||||
{
|
||||
"type": "file",
|
||||
"file": {
|
||||
"file_id": "gs://bucket/video.mp4",
|
||||
"format": "video/mp4",
|
||||
"video_metadata": video_metadata,
|
||||
},
|
||||
},
|
||||
],
|
||||
}
|
||||
]
|
||||
|
||||
def _get_file_part(self, contents: list) -> dict:
|
||||
for part in contents[0]["parts"]:
|
||||
if "file_data" in part:
|
||||
return part
|
||||
raise AssertionError("No file part found in contents")
|
||||
|
||||
def test_video_metadata_fps_gemini_2_5_flash(self):
|
||||
"""Gemini 2.5 Flash: fps in video_metadata should be forwarded (Issue #25474)"""
|
||||
messages = self._make_video_messages({"fps": 5})
|
||||
contents = _gemini_convert_messages_with_history(
|
||||
messages=messages, model="gemini-2.5-flash"
|
||||
)
|
||||
file_part = self._get_file_part(contents)
|
||||
assert "video_metadata" in file_part
|
||||
assert file_part["video_metadata"]["fps"] == 5
|
||||
|
||||
def test_video_metadata_fps_gemini_2_5_pro(self):
|
||||
"""Gemini 2.5 Pro: fps in video_metadata should be forwarded (Issue #25474)"""
|
||||
messages = self._make_video_messages({"fps": 10})
|
||||
contents = _gemini_convert_messages_with_history(
|
||||
messages=messages, model="gemini-2.5-pro"
|
||||
)
|
||||
file_part = self._get_file_part(contents)
|
||||
assert "video_metadata" in file_part
|
||||
assert file_part["video_metadata"]["fps"] == 10
|
||||
|
||||
def test_video_metadata_offsets_gemini_2_5_flash(self):
|
||||
"""Gemini 2.5 Flash: start_offset/end_offset converted to camelCase (Issue #25474)"""
|
||||
messages = self._make_video_messages(
|
||||
{"start_offset": "5s", "end_offset": "30s"}
|
||||
)
|
||||
contents = _gemini_convert_messages_with_history(
|
||||
messages=messages, model="gemini-2.5-flash"
|
||||
)
|
||||
file_part = self._get_file_part(contents)
|
||||
assert "video_metadata" in file_part
|
||||
vm = file_part["video_metadata"]
|
||||
assert vm["startOffset"] == "5s"
|
||||
assert vm["endOffset"] == "30s"
|
||||
|
||||
def test_video_metadata_all_fields_gemini_2_5_flash(self):
|
||||
"""Gemini 2.5 Flash: all video_metadata fields forwarded correctly (Issue #25474)"""
|
||||
messages = self._make_video_messages(
|
||||
{"fps": 5, "start_offset": "10s", "end_offset": "60s"}
|
||||
)
|
||||
contents = _gemini_convert_messages_with_history(
|
||||
messages=messages, model="gemini-2.5-flash"
|
||||
)
|
||||
file_part = self._get_file_part(contents)
|
||||
assert "video_metadata" in file_part
|
||||
vm = file_part["video_metadata"]
|
||||
assert vm["fps"] == 5
|
||||
assert vm["startOffset"] == "10s"
|
||||
assert vm["endOffset"] == "60s"
|
||||
|
||||
def test_video_metadata_gemini_1_5_pro(self):
|
||||
"""Gemini 1.5 Pro: video_metadata should also be forwarded (Issue #25474)"""
|
||||
messages = self._make_video_messages({"fps": 2})
|
||||
contents = _gemini_convert_messages_with_history(
|
||||
messages=messages, model="gemini-1.5-pro"
|
||||
)
|
||||
file_part = self._get_file_part(contents)
|
||||
assert "video_metadata" in file_part
|
||||
assert file_part["video_metadata"]["fps"] == 2
|
||||
|
||||
|
||||
def test_convert_tool_response_with_base64_image():
|
||||
"""Test tool response with base64 data URI image."""
|
||||
# Create a small test image (1x1 red pixel PNG)
|
||||
|
|
|
|||
|
|
@ -3476,8 +3476,8 @@ def test_new_detail_levels():
|
|||
assert file_part["media_resolution"] == {"level": "MEDIA_RESOLUTION_MEDIUM"}
|
||||
|
||||
|
||||
def test_video_metadata_only_for_gemini_3():
|
||||
"""Test that video_metadata is only applied for Gemini 3+ models (Issue #19026)"""
|
||||
def test_video_metadata_supported_for_all_gemini_models():
|
||||
"""Test that video_metadata is applied for all Gemini models (Issue #25474)"""
|
||||
from litellm.llms.vertex_ai.gemini.transformation import (
|
||||
_gemini_convert_messages_with_history,
|
||||
)
|
||||
|
|
@ -3499,39 +3499,29 @@ def test_video_metadata_only_for_gemini_3():
|
|||
}
|
||||
]
|
||||
|
||||
# Test with Gemini 1.5 (should not have video_metadata or media_resolution)
|
||||
contents_1_5 = _gemini_convert_messages_with_history(
|
||||
messages=messages, model="gemini-1.5-pro"
|
||||
)
|
||||
for model in ["gemini-1.5-pro", "gemini-2.5-flash", "gemini-2.5-pro", "gemini-3-pro-preview"]:
|
||||
contents = _gemini_convert_messages_with_history(messages=messages, model=model)
|
||||
|
||||
file_part_1_5 = None
|
||||
for part in contents_1_5[0]["parts"]:
|
||||
if "file_data" in part:
|
||||
file_part_1_5 = part
|
||||
break
|
||||
file_part = None
|
||||
for part in contents[0]["parts"]:
|
||||
if "file_data" in part:
|
||||
file_part = part
|
||||
break
|
||||
|
||||
assert file_part_1_5 is not None
|
||||
assert (
|
||||
"media_resolution" not in file_part_1_5
|
||||
), "Gemini 1.5 should not have media_resolution"
|
||||
assert (
|
||||
"video_metadata" not in file_part_1_5
|
||||
), "Gemini 1.5 should not have video_metadata"
|
||||
assert file_part is not None, f"{model}: file part should exist"
|
||||
assert "video_metadata" in file_part, f"{model}: video_metadata should be present"
|
||||
assert file_part["video_metadata"]["fps"] == 5, f"{model}: fps should be 5"
|
||||
|
||||
# Test with Gemini 3 (should have both)
|
||||
contents_3 = _gemini_convert_messages_with_history(
|
||||
messages=messages, model="gemini-3-pro-preview"
|
||||
)
|
||||
# Per-part media_resolution is Gemini 3+ only; 2.x uses generation_config global
|
||||
for model in ["gemini-3-pro-preview"]:
|
||||
contents = _gemini_convert_messages_with_history(messages=messages, model=model)
|
||||
file_part = next(p for p in contents[0]["parts"] if "file_data" in p)
|
||||
assert "media_resolution" in file_part, f"{model}: media_resolution should be present"
|
||||
|
||||
file_part_3 = None
|
||||
for part in contents_3[0]["parts"]:
|
||||
if "file_data" in part:
|
||||
file_part_3 = part
|
||||
break
|
||||
|
||||
assert file_part_3 is not None
|
||||
assert "media_resolution" in file_part_3, "Gemini 3 should have media_resolution"
|
||||
assert "video_metadata" in file_part_3, "Gemini 3 should have video_metadata"
|
||||
for model in ["gemini-1.5-pro", "gemini-2.5-flash", "gemini-2.5-pro"]:
|
||||
contents = _gemini_convert_messages_with_history(messages=messages, model=model)
|
||||
file_part = next(p for p in contents[0]["parts"] if "file_data" in p)
|
||||
assert "media_resolution" not in file_part, f"{model}: per-part media_resolution should not be set"
|
||||
|
||||
|
||||
def test_chunk_parser_handles_prompt_feedback_block():
|
||||
|
|
|
|||
|
|
@ -450,3 +450,193 @@ async def test_semantic_filter_hook_skips_no_tools():
|
|||
# Should return None (no modification)
|
||||
assert result is None, "Hook should skip requests without tools"
|
||||
print("✅ Hook correctly skips requests without tools")
|
||||
|
||||
|
||||
class TestGetToolsByNames:
|
||||
"""
|
||||
Regression coverage for SemanticMCPToolFilter._get_tools_by_names
|
||||
name-matching behavior (issue #26078).
|
||||
|
||||
The canonical name stored in the router is what the proxy's MCP
|
||||
registry emits (e.g. ``fc_web_search-firecrawl_scrape``). Some MCP
|
||||
clients — notably opencode — wrap every tool name with their own
|
||||
additive namespace prefix before sending it back in ``tools[]``, so
|
||||
the incoming name is ``litellm_fc_web_search-firecrawl_scrape``.
|
||||
|
||||
Exact-equality matching against the canonical dropped every such
|
||||
tool, the proxy forwarded ``tools: []`` with ``tool_choice: auto``,
|
||||
and strict upstream providers returned 400.
|
||||
"""
|
||||
|
||||
def _make_filter(self):
|
||||
from litellm.proxy._experimental.mcp_server.semantic_tool_filter import (
|
||||
SemanticMCPToolFilter,
|
||||
)
|
||||
|
||||
return SemanticMCPToolFilter(
|
||||
embedding_model="text-embedding-3-small",
|
||||
litellm_router_instance=Mock(),
|
||||
top_k=5,
|
||||
similarity_threshold=0.3,
|
||||
enabled=True,
|
||||
)
|
||||
|
||||
def test_exact_match_unchanged(self):
|
||||
"""Incoming name equals canonical — the historical path still works."""
|
||||
filter_instance = self._make_filter()
|
||||
available_tools = [
|
||||
{"name": "get_weather", "description": "fetch weather"},
|
||||
{"name": "send_email", "description": "send mail"},
|
||||
]
|
||||
|
||||
matched = filter_instance._get_tools_by_names(
|
||||
["send_email"], available_tools
|
||||
)
|
||||
|
||||
assert len(matched) == 1
|
||||
assert matched[0]["name"] == "send_email"
|
||||
|
||||
def test_client_prefix_with_underscore_separator(self):
|
||||
"""Client wraps canonical with ``<alias>_`` (opencode pattern)."""
|
||||
filter_instance = self._make_filter()
|
||||
canonical = "fc_web_search-firecrawl_scrape"
|
||||
client_name = "litellm_" + canonical
|
||||
available_tools = [{"name": client_name, "description": "scrape"}]
|
||||
|
||||
matched = filter_instance._get_tools_by_names(
|
||||
[canonical], available_tools
|
||||
)
|
||||
|
||||
assert len(matched) == 1
|
||||
# Must return the incoming tool unchanged so the client-facing
|
||||
# name survives, otherwise tool-call round-trips break client-side.
|
||||
assert matched[0]["name"] == client_name
|
||||
|
||||
def test_client_prefix_with_dash_separator(self):
|
||||
"""Some clients use dash as alias separator; accept that too."""
|
||||
filter_instance = self._make_filter()
|
||||
canonical = "weather_svc-get_weather"
|
||||
available_tools = [
|
||||
{"name": "mcp-" + canonical, "description": "weather"}
|
||||
]
|
||||
|
||||
matched = filter_instance._get_tools_by_names(
|
||||
[canonical], available_tools
|
||||
)
|
||||
|
||||
assert len(matched) == 1
|
||||
assert matched[0]["name"] == "mcp-" + canonical
|
||||
|
||||
def test_suffix_without_separator_does_not_match(self):
|
||||
"""
|
||||
A bare-substring suffix must not match — ``rain_gear`` is not a
|
||||
namespaced version of canonical ``ear`` and the user would be
|
||||
surprised to see it selected.
|
||||
"""
|
||||
filter_instance = self._make_filter()
|
||||
available_tools = [{"name": "rain_gear", "description": "raincoat"}]
|
||||
|
||||
matched = filter_instance._get_tools_by_names(["ear"], available_tools)
|
||||
|
||||
assert matched == []
|
||||
|
||||
def test_exact_match_preferred_over_prefixed(self):
|
||||
"""
|
||||
When both a bare canonical and a client-prefixed variant are
|
||||
present, the bare one wins so ordering is stable.
|
||||
"""
|
||||
filter_instance = self._make_filter()
|
||||
canonical = "search"
|
||||
available_tools = [
|
||||
{"name": canonical, "description": "plain"},
|
||||
{"name": "litellm_" + canonical, "description": "wrapped"},
|
||||
]
|
||||
|
||||
matched = filter_instance._get_tools_by_names(
|
||||
[canonical], available_tools
|
||||
)
|
||||
|
||||
assert len(matched) == 1
|
||||
assert matched[0]["name"] == canonical
|
||||
|
||||
def test_same_tool_not_returned_twice(self):
|
||||
"""
|
||||
Two distinct canonicals that both suffix-match the same incoming
|
||||
tool must not produce a duplicate in the output list.
|
||||
``fs-read_file`` and ``api-fs-read_file`` are both valid
|
||||
separator-anchored suffixes of ``litellm_api-fs-read_file``.
|
||||
"""
|
||||
filter_instance = self._make_filter()
|
||||
available_tools = [
|
||||
{"name": "litellm_api-fs-read_file", "description": "read"}
|
||||
]
|
||||
|
||||
matched = filter_instance._get_tools_by_names(
|
||||
["fs-read_file", "api-fs-read_file"], available_tools
|
||||
)
|
||||
|
||||
assert len(matched) == 1
|
||||
|
||||
def test_suffix_fallback_prefers_shortest_candidate(self):
|
||||
"""
|
||||
When no exact match exists and several incoming tools
|
||||
suffix-match the same canonical, the one closest in length to
|
||||
the canonical (i.e. the least-wrapped) should be chosen.
|
||||
"""
|
||||
filter_instance = self._make_filter()
|
||||
canonical = "svc-search"
|
||||
available_tools = [
|
||||
{"name": "my_tag_" + canonical, "description": "tag search"},
|
||||
{"name": "my_" + canonical, "description": "plain search"},
|
||||
]
|
||||
|
||||
matched = filter_instance._get_tools_by_names(
|
||||
[canonical], available_tools
|
||||
)
|
||||
|
||||
assert len(matched) == 1
|
||||
assert matched[0]["name"] == "my_" + canonical
|
||||
|
||||
def test_ordering_follows_router_output(self):
|
||||
"""Returned tools follow the order the semantic router chose."""
|
||||
filter_instance = self._make_filter()
|
||||
available_tools = [
|
||||
{"name": "litellm_fs-read", "description": "read"},
|
||||
{"name": "litellm_fs-write", "description": "write"},
|
||||
{"name": "litellm_fs-delete", "description": "delete"},
|
||||
]
|
||||
|
||||
matched = filter_instance._get_tools_by_names(
|
||||
["fs-write", "fs-delete", "fs-read"], available_tools
|
||||
)
|
||||
|
||||
names = [t["name"] for t in matched]
|
||||
assert names == [
|
||||
"litellm_fs-write",
|
||||
"litellm_fs-delete",
|
||||
"litellm_fs-read",
|
||||
]
|
||||
|
||||
def test_does_not_collide_with_local_function_on_unprefixed_canonical(self):
|
||||
"""
|
||||
Guard against the collision @krrish-berri-2 flagged on #26117:
|
||||
if the canonical name from the router is not server-prefixed
|
||||
(i.e. does not contain ``MCP_TOOL_PREFIX_SEPARATOR``), suffix
|
||||
matching must not kick in. Otherwise an unrelated local user
|
||||
function whose name happens to end in the canonical substring
|
||||
would be spuriously selected.
|
||||
"""
|
||||
filter_instance = self._make_filter()
|
||||
available_tools = [
|
||||
{
|
||||
"name": "my_firecrawl_scrape",
|
||||
"description": "unrelated local function",
|
||||
},
|
||||
]
|
||||
|
||||
matched = filter_instance._get_tools_by_names(
|
||||
["firecrawl_scrape"], # no MCP_TOOL_PREFIX_SEPARATOR in canonical
|
||||
available_tools,
|
||||
)
|
||||
|
||||
assert matched == []
|
||||
|
|
|
|||
401
tests/test_litellm/test_dashscope_image_generation.py
Normal file
401
tests/test_litellm/test_dashscope_image_generation.py
Normal file
|
|
@ -0,0 +1,401 @@
|
|||
"""
|
||||
Unit tests for DashScope image generation support (qwen-image-2.0, qwen-image-2.0-pro).
|
||||
|
||||
Run in docker: pytest tests/test_litellm/test_dashscope_image_generation.py -v
|
||||
"""
|
||||
|
||||
import json
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import httpx
|
||||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.llms.dashscope.image_generation.transformation import (
|
||||
DashScopeImageGenerationConfig,
|
||||
DEFAULT_API_BASE,
|
||||
)
|
||||
from litellm.types.utils import ImageObject, ImageResponse
|
||||
from litellm.utils import get_llm_provider
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 1. Provider detection
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model_string",
|
||||
[
|
||||
"dashscope/qwen-image-2.0",
|
||||
"dashscope/qwen-image-2.0-pro",
|
||||
],
|
||||
)
|
||||
def test_get_llm_provider_returns_dashscope(model_string: str):
|
||||
model, provider, _, _ = get_llm_provider(model_string)
|
||||
assert provider == "dashscope", f"Expected 'dashscope', got '{provider}'"
|
||||
assert "qwen-image" in model
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 2. Model info: mode == "image_generation"
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model_string, custom_provider",
|
||||
[
|
||||
("dashscope/qwen-image-2.0", "dashscope"),
|
||||
("dashscope/qwen-image-2.0-pro", "dashscope"),
|
||||
],
|
||||
)
|
||||
def test_get_model_info_mode_is_image_generation(
|
||||
model_string: str, custom_provider: str
|
||||
):
|
||||
import os
|
||||
|
||||
prev_env = os.environ.get("LITELLM_LOCAL_MODEL_COST_MAP")
|
||||
prev_model_cost = litellm.model_cost
|
||||
try:
|
||||
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
|
||||
litellm.model_cost = litellm.get_model_cost_map(url="")
|
||||
|
||||
info = litellm.get_model_info(
|
||||
model=model_string, custom_llm_provider=custom_provider
|
||||
)
|
||||
assert (
|
||||
info["mode"] == "image_generation"
|
||||
), f"Expected mode='image_generation', got '{info['mode']}'"
|
||||
finally:
|
||||
if prev_env is None:
|
||||
os.environ.pop("LITELLM_LOCAL_MODEL_COST_MAP", None)
|
||||
else:
|
||||
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = prev_env
|
||||
litellm.model_cost = prev_model_cost
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 3. Request transformation
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestDashScopeImageGenerationConfig:
|
||||
def setup_method(self):
|
||||
self.cfg = DashScopeImageGenerationConfig()
|
||||
|
||||
def test_get_complete_url_default(self):
|
||||
url = self.cfg.get_complete_url(None, None, "qwen-image-2.0", {}, {})
|
||||
assert url == DEFAULT_API_BASE
|
||||
|
||||
def test_get_complete_url_custom(self):
|
||||
custom = "https://custom.endpoint/generate"
|
||||
url = self.cfg.get_complete_url(custom, None, "qwen-image-2.0", {}, {})
|
||||
assert url == custom
|
||||
|
||||
def test_validate_environment_sets_auth_header(self):
|
||||
headers = self.cfg.validate_environment(
|
||||
headers={},
|
||||
model="qwen-image-2.0",
|
||||
messages=[],
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
api_key="sk-test-key",
|
||||
)
|
||||
assert headers["Authorization"] == "Bearer sk-test-key"
|
||||
assert headers["Content-Type"] == "application/json"
|
||||
|
||||
def test_validate_environment_raises_without_key(self):
|
||||
with patch(
|
||||
"litellm.llms.dashscope.image_generation.transformation.get_secret_str",
|
||||
return_value=None,
|
||||
):
|
||||
with pytest.raises(ValueError, match="DASHSCOPE_API_KEY"):
|
||||
self.cfg.validate_environment(
|
||||
headers={},
|
||||
model="qwen-image-2.0",
|
||||
messages=[],
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
api_key=None,
|
||||
)
|
||||
|
||||
def test_transform_request_structure(self):
|
||||
req = self.cfg.transform_image_generation_request(
|
||||
model="qwen-image-2.0",
|
||||
prompt="a puppy on green grass",
|
||||
optional_params={"size": "1024*1024"},
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
assert req["model"] == "qwen-image-2.0"
|
||||
messages = req["input"]["messages"]
|
||||
assert len(messages) == 1
|
||||
assert messages[0]["role"] == "user"
|
||||
assert messages[0]["content"][0]["text"] == "a puppy on green grass"
|
||||
assert req["parameters"]["size"] == "1024*1024"
|
||||
|
||||
def test_transform_request_empty_params(self):
|
||||
req = self.cfg.transform_image_generation_request(
|
||||
model="qwen-image-2.0-pro",
|
||||
prompt="sunset over the ocean",
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
assert req["parameters"] == {}
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 4. Response transformation
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _make_mock_response(self, image_url: str) -> httpx.Response:
|
||||
body = {
|
||||
"status_code": 200,
|
||||
"request_id": "test-request-id",
|
||||
"output": {
|
||||
"choices": [
|
||||
{
|
||||
"finish_reason": "stop",
|
||||
"message": {
|
||||
"role": "assistant",
|
||||
"content": [{"image": image_url}],
|
||||
},
|
||||
}
|
||||
]
|
||||
},
|
||||
"usage": {
|
||||
"input_tokens": 0,
|
||||
"output_tokens": 0,
|
||||
"width": 1024,
|
||||
"height": 1024,
|
||||
"image_count": 1,
|
||||
},
|
||||
}
|
||||
mock_resp = MagicMock(spec=httpx.Response)
|
||||
mock_resp.status_code = 200
|
||||
mock_resp.headers = {}
|
||||
mock_resp.json.return_value = body
|
||||
return mock_resp
|
||||
|
||||
def test_transform_response_extracts_url(self):
|
||||
image_url = "https://example.oss.aliyuncs.com/generated/test.png"
|
||||
mock_resp = self._make_mock_response(image_url)
|
||||
model_response = ImageResponse()
|
||||
result = self.cfg.transform_image_generation_response(
|
||||
model="qwen-image-2.0",
|
||||
raw_response=mock_resp,
|
||||
model_response=model_response,
|
||||
logging_obj=MagicMock(),
|
||||
request_data={},
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
encoding=None,
|
||||
)
|
||||
assert result.data is not None
|
||||
assert len(result.data) == 1
|
||||
assert result.data[0].url == image_url
|
||||
|
||||
def test_transform_response_multiple_images(self):
|
||||
body = {
|
||||
"output": {
|
||||
"choices": [
|
||||
{
|
||||
"finish_reason": "stop",
|
||||
"message": {
|
||||
"role": "assistant",
|
||||
"content": [{"image": "https://example.com/img1.png"}],
|
||||
},
|
||||
},
|
||||
{
|
||||
"finish_reason": "stop",
|
||||
"message": {
|
||||
"role": "assistant",
|
||||
"content": [{"image": "https://example.com/img2.png"}],
|
||||
},
|
||||
},
|
||||
]
|
||||
},
|
||||
"usage": {},
|
||||
}
|
||||
mock_resp = MagicMock(spec=httpx.Response)
|
||||
mock_resp.status_code = 200
|
||||
mock_resp.headers = {}
|
||||
mock_resp.json.return_value = body
|
||||
|
||||
model_response = ImageResponse()
|
||||
result = self.cfg.transform_image_generation_response(
|
||||
model="qwen-image-2.0",
|
||||
raw_response=mock_resp,
|
||||
model_response=model_response,
|
||||
logging_obj=MagicMock(),
|
||||
request_data={},
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
encoding=None,
|
||||
)
|
||||
assert len(result.data) == 2
|
||||
assert result.data[0].url == "https://example.com/img1.png"
|
||||
assert result.data[1].url == "https://example.com/img2.png"
|
||||
|
||||
def test_transform_response_raises_on_non_200_status(self):
|
||||
mock_resp = MagicMock(spec=httpx.Response)
|
||||
mock_resp.status_code = 400
|
||||
mock_resp.headers = {}
|
||||
mock_resp.text = '{"code":"InvalidParameter","message":"Size not supported"}'
|
||||
mock_resp.json.return_value = {
|
||||
"code": "InvalidParameter",
|
||||
"message": "Size not supported",
|
||||
}
|
||||
|
||||
with pytest.raises(Exception):
|
||||
self.cfg.transform_image_generation_response(
|
||||
model="qwen-image-2.0",
|
||||
raw_response=mock_resp,
|
||||
model_response=ImageResponse(),
|
||||
logging_obj=MagicMock(),
|
||||
request_data={},
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
encoding=None,
|
||||
)
|
||||
|
||||
def test_transform_response_raises_on_api_error_body(self):
|
||||
mock_resp = MagicMock(spec=httpx.Response)
|
||||
mock_resp.status_code = 200
|
||||
mock_resp.headers = {}
|
||||
mock_resp.json.return_value = {
|
||||
"code": "InvalidParameter",
|
||||
"message": "Size not supported",
|
||||
}
|
||||
|
||||
with pytest.raises(Exception):
|
||||
self.cfg.transform_image_generation_response(
|
||||
model="qwen-image-2.0",
|
||||
raw_response=mock_resp,
|
||||
model_response=ImageResponse(),
|
||||
logging_obj=MagicMock(),
|
||||
request_data={},
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
encoding=None,
|
||||
)
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 5. OpenAI → DashScope parameter mapping
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_map_openai_params_size_conversion(self):
|
||||
mapped = self.cfg.map_openai_params(
|
||||
non_default_params={"size": "1024x1024"},
|
||||
optional_params={},
|
||||
model="qwen-image-2.0",
|
||||
drop_params=False,
|
||||
)
|
||||
assert mapped["size"] == "1024*1024"
|
||||
|
||||
def test_map_openai_params_n_to_image_count(self):
|
||||
mapped = self.cfg.map_openai_params(
|
||||
non_default_params={"n": 2},
|
||||
optional_params={},
|
||||
model="qwen-image-2.0",
|
||||
drop_params=False,
|
||||
)
|
||||
assert mapped["image_count"] == 2
|
||||
|
||||
def test_map_openai_params_unknown_size_uses_asterisk(self):
|
||||
mapped = self.cfg.map_openai_params(
|
||||
non_default_params={"size": "768x768"},
|
||||
optional_params={},
|
||||
model="qwen-image-2.0",
|
||||
drop_params=False,
|
||||
)
|
||||
assert mapped["size"] == "768*768"
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"openai_size, expected",
|
||||
[
|
||||
("256x256", "256*256"),
|
||||
("512x512", "512*512"),
|
||||
("1024x1024", "1024*1024"),
|
||||
("1792x1024", "1792*1024"),
|
||||
("1024x1792", "1024*1792"),
|
||||
("2048x2048", "2048*2048"),
|
||||
],
|
||||
)
|
||||
def test_map_openai_params_size_table(self, openai_size: str, expected: str):
|
||||
mapped = self.cfg.map_openai_params(
|
||||
non_default_params={"size": openai_size},
|
||||
optional_params={},
|
||||
model="qwen-image-2.0",
|
||||
drop_params=False,
|
||||
)
|
||||
assert mapped["size"] == expected
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 6. End-to-end flow via litellm.image_generation (HTTP mocked)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_litellm_image_generation_dashscope_end_to_end():
|
||||
mock_response_body = {
|
||||
"output": {
|
||||
"choices": [
|
||||
{
|
||||
"finish_reason": "stop",
|
||||
"message": {
|
||||
"role": "assistant",
|
||||
"content": [
|
||||
{
|
||||
"image": "https://dashscope-result.oss.aliyuncs.com/test.png"
|
||||
}
|
||||
],
|
||||
},
|
||||
}
|
||||
]
|
||||
},
|
||||
"usage": {
|
||||
"input_tokens": 0,
|
||||
"output_tokens": 0,
|
||||
"width": 1024,
|
||||
"height": 1024,
|
||||
"image_count": 1,
|
||||
},
|
||||
}
|
||||
|
||||
with patch(
|
||||
"litellm.llms.custom_httpx.llm_http_handler.HTTPHandler.post"
|
||||
) as mock_post:
|
||||
mock_http_response = MagicMock()
|
||||
mock_http_response.json.return_value = mock_response_body
|
||||
mock_http_response.status_code = 200
|
||||
mock_http_response.headers = {}
|
||||
mock_post.return_value = mock_http_response
|
||||
|
||||
response = litellm.image_generation(
|
||||
model="dashscope/qwen-image-2.0",
|
||||
prompt="a puppy playing on green grass",
|
||||
api_key="sk-test-key",
|
||||
size="1024x1024",
|
||||
)
|
||||
|
||||
assert response is not None
|
||||
assert response.data is not None
|
||||
assert len(response.data) == 1
|
||||
assert (
|
||||
response.data[0].url == "https://dashscope-result.oss.aliyuncs.com/test.png"
|
||||
)
|
||||
|
||||
# Verify the HTTP call was made to the DashScope endpoint
|
||||
call_args = mock_post.call_args
|
||||
called_url = (
|
||||
call_args[0][0] if call_args[0] else call_args.kwargs.get("url", "")
|
||||
)
|
||||
assert "dashscope" in called_url or "aliyuncs" in called_url
|
||||
|
||||
# Verify request body contains DashScope format
|
||||
call_kwargs = call_args[1] if call_args[1] else {}
|
||||
if "json" in call_kwargs:
|
||||
body = call_kwargs["json"]
|
||||
assert "input" in body
|
||||
assert "messages" in body["input"]
|
||||
|
|
@ -243,6 +243,7 @@ export default function SpendLogsTable({
|
|||
allTeams,
|
||||
handleFilterChange,
|
||||
handleFilterReset: handleFilterResetFromHook,
|
||||
refetchWithFilters,
|
||||
} = useLogFilterLogic({
|
||||
logs: logsData,
|
||||
accessToken,
|
||||
|
|
@ -363,7 +364,14 @@ export default function SpendLogsTable({
|
|||
|
||||
// Add this function to handle manual refresh
|
||||
const handleRefresh = () => {
|
||||
logs.refetch();
|
||||
if (hasBackendFilters) {
|
||||
// When backend filters (e.g. Key Alias) are active the main TanStack Query
|
||||
// is disabled and its params do not include filter values like key_alias.
|
||||
// Route through the filter-aware refetch so all active filters are preserved.
|
||||
refetchWithFilters();
|
||||
} else {
|
||||
logs.refetch();
|
||||
}
|
||||
};
|
||||
|
||||
const handleRowClick = (log: LogEntry) => {
|
||||
|
|
|
|||
|
|
@ -71,6 +71,14 @@ export function useLogFilterLogic({
|
|||
const [filters, setFilters] = useState<LogFilterState>(defaultFilters);
|
||||
const [backendFilteredLogs, setBackendFilteredLogs] = useState<PaginatedResponse | null>(null);
|
||||
const lastSearchTimestamp = useRef(0);
|
||||
|
||||
// Refs that always hold the latest filters and hasBackendFilters values.
|
||||
// The sort/page/time effect below intentionally omits these from its dep array
|
||||
// to avoid double-fetches when a filter changes; reading from refs instead of
|
||||
// the closure prevents stale-closure bugs (e.g. the effect using a snapshot of
|
||||
// filters taken before the user selected Key Alias).
|
||||
const filtersRef = useRef(filters);
|
||||
const hasBackendFiltersRef = useRef(false);
|
||||
const performSearch = useCallback(
|
||||
async (filters: LogFilterState, page = 1) => {
|
||||
if (!accessToken) return;
|
||||
|
|
@ -152,18 +160,25 @@ export function useLogFilterLogic({
|
|||
[filters],
|
||||
);
|
||||
|
||||
// Keep refs in sync on every render so the sort/page/time effect always reads
|
||||
// the latest values without those values being in its dep array.
|
||||
useEffect(() => {
|
||||
filtersRef.current = filters;
|
||||
hasBackendFiltersRef.current = hasBackendFilters;
|
||||
}, [filters, hasBackendFilters]);
|
||||
|
||||
// Refetch when sort, page, or time range changes (backend filters use their own fetch, not the main query)
|
||||
useEffect(() => {
|
||||
if (hasBackendFilters && accessToken) {
|
||||
if (hasBackendFiltersRef.current && accessToken) {
|
||||
// Cancel any pending debounced search to prevent it from overwriting this page's results
|
||||
debouncedSearch.cancel();
|
||||
performSearch(filters, currentPage);
|
||||
performSearch(filtersRef.current, currentPage);
|
||||
}
|
||||
// Intentionally omitted from deps:
|
||||
// - `filters` / `debouncedSearch` / `performSearch`: filter changes are handled by
|
||||
// handleFilterChange → debouncedSearch; adding them here would double-fetch on filter apply.
|
||||
// - `hasBackendFilters` / `accessToken`: stable across sort/page/time changes; including them
|
||||
// would cause spurious re-runs when the filter state first becomes active.
|
||||
// filters / hasBackendFilters are read via refs — avoids stale-closure bugs
|
||||
// when sort/page/time changes after a filter (e.g. Key Alias) was set.
|
||||
// debouncedSearch / performSearch: filter changes go through handleFilterChange
|
||||
// → debouncedSearch; adding them here would cause double-fetches on filter apply.
|
||||
// accessToken: stable across sort/page/time changes.
|
||||
// eslint-disable-next-line react-hooks/exhaustive-deps
|
||||
}, [sortBy, sortOrder, currentPage, startTime, endTime, isCustomDate]);
|
||||
|
||||
|
|
@ -299,6 +314,20 @@ export function useLogFilterLogic({
|
|||
setCurrentPage(1);
|
||||
};
|
||||
|
||||
// Expose a filter-aware refetch so callers (e.g. the manual Fetch button) can
|
||||
// refresh results while keeping all active backend filters intact. The plain
|
||||
// `logs.refetch()` in the parent only re-runs the main TanStack Query, which
|
||||
// does not carry key_alias or other backend-only filter params.
|
||||
const refetchWithFilters = useCallback(
|
||||
(page = currentPage) => {
|
||||
if (hasBackendFilters && accessToken) {
|
||||
debouncedSearch.cancel();
|
||||
performSearch(filters, page);
|
||||
}
|
||||
},
|
||||
[hasBackendFilters, accessToken, filters, currentPage, performSearch, debouncedSearch],
|
||||
);
|
||||
|
||||
return {
|
||||
filters,
|
||||
filteredLogs,
|
||||
|
|
@ -306,5 +335,6 @@ export function useLogFilterLogic({
|
|||
allTeams,
|
||||
handleFilterChange,
|
||||
handleFilterReset,
|
||||
refetchWithFilters,
|
||||
};
|
||||
}
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue