diff --git a/litellm/llms/dashscope/common_utils.py b/litellm/llms/dashscope/common_utils.py index 952960e8207..3e2e718a469 100644 --- a/litellm/llms/dashscope/common_utils.py +++ b/litellm/llms/dashscope/common_utils.py @@ -16,6 +16,7 @@ if TYPE_CHECKING: BaseImageGenerationConfig, ) from litellm.llms.base_llm.rerank.transformation import BaseRerankConfig + from litellm.llms.base_llm.videos.transformation import BaseVideoConfig DASHSCOPE_CHAT_COMPATIBLE_PATH: Final = "/compatible-mode/v1" DASHSCOPE_RERANK_PATH: Final = "/compatible-api/v1/reranks" @@ -89,6 +90,22 @@ def get_dashscope_family_image_generation_config( return DashScopeImageGenerationConfig() +def get_dashscope_family_video_config( + custom_llm_provider: str, +) -> "BaseVideoConfig": + if custom_llm_provider == "qwencloud": + from litellm.llms.dashscope.qwencloud import QwenCloudVideoConfig + + return QwenCloudVideoConfig() + if custom_llm_provider == "qwen_ai_platform": + from litellm.llms.dashscope.qwen_ai_platform import QwenAIPlatformVideoConfig + + return QwenAIPlatformVideoConfig() + from litellm.llms.dashscope.videos.transformation import DashScopeVideoConfig + + return DashScopeVideoConfig() + + def resolve_dashscope_family_api_key(custom_llm_provider: str, api_key: str | None) -> str | None: if custom_llm_provider == "dashscope": return api_key or get_secret_str("DASHSCOPE_API_KEY") diff --git a/litellm/llms/dashscope/qwen_ai_platform.py b/litellm/llms/dashscope/qwen_ai_platform.py index 864f0f459e4..f7cc50cc82a 100644 --- a/litellm/llms/dashscope/qwen_ai_platform.py +++ b/litellm/llms/dashscope/qwen_ai_platform.py @@ -7,12 +7,14 @@ from .common_utils import resolve_dashscope_family_rerank_api_base from .embed.transformation import DashScopeEmbeddingConfig from .image_generation.transformation import DashScopeImageGenerationConfig from .rerank.transformation import DashScopeRerankConfig +from .videos.transformation import DashScopeVideoConfig QWEN_AI_PLATFORM_API_BASE: Final = "https://dashscope.aliyuncs.com/compatible-mode/v1" QWEN_AI_PLATFORM_RERANK_API_BASE: Final = "https://dashscope.aliyuncs.com/compatible-api/v1/reranks" QWEN_AI_PLATFORM_IMAGE_API_BASE: Final = ( "https://dashscope.aliyuncs.com/api/v1/services/aigc/multimodal-generation/generation" ) +QWEN_AI_PLATFORM_VIDEO_API_BASE: Final = "https://dashscope.aliyuncs.com" def _resolve_qwen_ai_platform_api_key(api_key: str | None) -> str | None: @@ -63,3 +65,11 @@ class QwenAIPlatformImageGenerationConfig(DashScopeImageGenerationConfig): def _resolve_image_api_base(self, image_api_base: str | None) -> str: return image_api_base or get_secret_str("QWEN_AI_PLATFORM_API_BASE_IMAGE") or QWEN_AI_PLATFORM_IMAGE_API_BASE + + +class QwenAIPlatformVideoConfig(DashScopeVideoConfig): + def _resolve_api_key(self, api_key: str | None) -> str: + return _require_qwen_ai_platform_api_key(api_key) + + def _resolve_video_api_base(self, video_api_base: str | None) -> str: + return video_api_base or get_secret_str("QWEN_AI_PLATFORM_API_BASE_VIDEO") or QWEN_AI_PLATFORM_VIDEO_API_BASE diff --git a/litellm/llms/dashscope/qwencloud.py b/litellm/llms/dashscope/qwencloud.py index ac21a48dd3e..a12e21b0983 100644 --- a/litellm/llms/dashscope/qwencloud.py +++ b/litellm/llms/dashscope/qwencloud.py @@ -7,12 +7,14 @@ from .common_utils import resolve_dashscope_family_rerank_api_base from .embed.transformation import DashScopeEmbeddingConfig from .image_generation.transformation import DashScopeImageGenerationConfig from .rerank.transformation import DashScopeRerankConfig +from .videos.transformation import DashScopeVideoConfig QWENCLOUD_API_BASE: Final = "https://dashscope-intl.aliyuncs.com/compatible-mode/v1" QWENCLOUD_RERANK_API_BASE: Final = "https://dashscope-intl.aliyuncs.com/compatible-api/v1/reranks" QWENCLOUD_IMAGE_API_BASE: Final = ( "https://dashscope-intl.aliyuncs.com/api/v1/services/aigc/multimodal-generation/generation" ) +QWENCLOUD_VIDEO_API_BASE: Final = "https://dashscope-intl.aliyuncs.com" def _resolve_qwencloud_api_key(api_key: str | None) -> str | None: @@ -63,3 +65,11 @@ class QwenCloudImageGenerationConfig(DashScopeImageGenerationConfig): def _resolve_image_api_base(self, image_api_base: str | None) -> str: return image_api_base or get_secret_str("QWENCLOUD_API_BASE_IMAGE") or QWENCLOUD_IMAGE_API_BASE + + +class QwenCloudVideoConfig(DashScopeVideoConfig): + def _resolve_api_key(self, api_key: str | None) -> str: + return _require_qwencloud_api_key(api_key) + + def _resolve_video_api_base(self, video_api_base: str | None) -> str: + return video_api_base or get_secret_str("QWENCLOUD_API_BASE_VIDEO") or QWENCLOUD_VIDEO_API_BASE diff --git a/litellm/llms/dashscope/videos/__init__.py b/litellm/llms/dashscope/videos/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/litellm/llms/dashscope/videos/transformation.py b/litellm/llms/dashscope/videos/transformation.py new file mode 100644 index 00000000000..81437906616 --- /dev/null +++ b/litellm/llms/dashscope/videos/transformation.py @@ -0,0 +1,706 @@ +""" +DashScope (Alibaba Model Studio) async video task API: create, poll and download. DashScope has no list, +delete or remix endpoint, so those raise. +""" + +import base64 +from collections.abc import Mapping +from datetime import datetime +from io import BufferedReader, BytesIO +from math import gcd +from types import MappingProxyType +from typing import TYPE_CHECKING, Final + +import httpx +from httpx._types import RequestFiles +from pydantic import BaseModel, TypeAdapter, ValidationError + +import litellm +from litellm.exceptions import UnsupportedParamsError +from litellm.images.utils import ImageEditRequestUtils +from litellm.litellm_core_utils.url_utils import encode_url_path_segment +from litellm.llms.base_llm.chat.transformation import BaseLLMException +from litellm.llms.base_llm.videos.transformation import BaseVideoConfig +from litellm.llms.custom_httpx.http_handler import ( + _get_httpx_client, # pyright: ignore[reportPrivateUsage, reportUnknownVariableType] # house cached-client factory has no public alias and its stub leaves params untyped + get_async_httpx_client, # pyright: ignore[reportUnknownVariableType] # factory stub leaves params untyped +) +from litellm.secret_managers.main import get_secret_str +from litellm.types.router import GenericLiteLLMParams +from litellm.types.videos.main import VideoCreateOptionalRequestParams, VideoObject +from litellm.types.videos.utils import ( + decode_video_id_with_provider, + encode_video_id_with_provider, + extract_original_video_id, +) +from litellm.utils import get_model_info + +if TYPE_CHECKING: + from litellm.litellm_core_utils.litellm_logging import Logging + from litellm.llms.custom_httpx.http_handler import HTTPHandler + + +class DashScopeVideoError(BaseLLMException): + pass + + +class _TaskOutput(BaseModel, frozen=True): + task_id: str = "" + task_status: str = "" + submit_time: str | None = None + scheduled_time: str | None = None + end_time: str | None = None + orig_prompt: str | None = None + video_url: str | None = None + code: str | None = None + message: str | None = None + + +class _TaskUsage(BaseModel, frozen=True): + video_count: int | None = None + duration: float | None = None + input_video_duration: float | None = None + output_video_duration: float | None = None + fps: int | None = None + SR: int | None = None + ratio: str | None = None + + +class _TaskResponse(BaseModel, frozen=True): + output: _TaskOutput | None = None + usage: _TaskUsage | None = None + request_id: str | None = None + code: str | None = None + message: str | None = None + + +DASHSCOPE_VIDEO_DEFAULT_API_BASE: Final = "https://dashscope.aliyuncs.com" +DASHSCOPE_VIDEO_SYNTHESIS_PATH: Final = "/api/v1/services/aigc/video-generation/video-synthesis" +DASHSCOPE_TASKS_PATH: Final = "/api/v1/tasks" +DASHSCOPE_COMPATIBLE_MODE_PATH: Final = "/compatible-mode/v1" + +_EMPTY_PARAMS: Final[dict[str, object]] = {} # mutable-ok: BaseVideoConfig contract empty params + +_TASK_RESPONSE_ADAPTER: Final = TypeAdapter(_TaskResponse) +_PARAMETERS_ADAPTER: Final = TypeAdapter(dict[str, object]) + +_STATUS_MAP: Final = MappingProxyType( + { + "PENDING": "queued", + "RUNNING": "in_progress", + "SUCCEEDED": "completed", + "FAILED": "failed", + "CANCELED": "cancelled", + "UNKNOWN": "failed", + } +) + +_PENDING_STATUSES: Final = frozenset({"PENDING", "RUNNING"}) + +_LEGACY_INPUT_KEYS: Final = ( + "img_url", + "first_frame_url", + "last_frame_url", + "audio_url", +) + +_INPUT_KEYS: Final = ("media", "negative_prompt", *_LEGACY_INPUT_KEYS) + +_REFERENCE_INPUT_FIELD: Final = "dashscope_video_reference_input" +_MEDIA_REFERENCE_TYPES: Final = frozenset({"first_frame", "reference_image"}) + +DASHSCOPE_DEFAULT_DURATION_SECONDS: Final = 5 +DASHSCOPE_DEFAULT_RESOLUTION: Final = "1080P" +DASHSCOPE_SMART_DURATION: Final = -1 + +_PARAMETER_KEYS: Final = ( + "resolution", + "ratio", + "duration", + "audio", + "seed", + "prompt_extend", + "watermark", +) + +_DROP_FROM_REQUEST: Final = frozenset( + { + "model", + "prompt", + "user", + "characters", + "image", + "extra_headers", + "extra_query", + "extra_body", + } +) + +_RESOLUTION_TIERS: Final = MappingProxyType({480: "480p", 720: "720p", 1080: "1080p"}) + +_RESOLUTION_HEIGHT_TIERS: Final[tuple[tuple[int, str], ...]] = ( + (600, "480p"), + (900, "720p"), +) + + +def _parse_task_response(raw_response: httpx.Response) -> _TaskResponse: + return _TASK_RESPONSE_ADAPTER.validate_python(raw_response.json()) + + +def _normalized_api_base(api_base: str) -> str: + trimmed: Final = api_base.rstrip("/") + if trimmed.endswith(DASHSCOPE_COMPATIBLE_MODE_PATH): + return trimmed[: -len(DASHSCOPE_COMPATIBLE_MODE_PATH)] + return trimmed + + +def _ratio_from_size(size: str) -> str | None: + if ":" in size: + return size + width, height = _size_dimensions(size) or (0, 0) + if not width or not height: + return None + divisor: Final = gcd(width, height) + return f"{width // divisor}:{height // divisor}" + + +def _size_dimensions(size: str) -> tuple[int, int] | None: + width_str, separator, height_str = size.partition("x") + if not separator or not (width_str.isdigit() and height_str.isdigit()): + return None + return int(width_str), int(height_str) + + +def _resolution_from_size(size: str) -> str | None: + """ + DashScope bills per second by named tier, so a size must map to its tier or it bills at the 1080P default. + """ + dimensions: Final = _size_dimensions(size) + if dimensions is None: + return None + shortest_side: Final = min(dimensions) + return next( + (label.upper() for threshold, label in _RESOLUTION_HEIGHT_TIERS if shortest_side < threshold), + "1080P", + ) + + +def _duration_param(seconds: object) -> int | None: + if isinstance(seconds, bool): + return None + if isinstance(seconds, int): + return seconds + if not isinstance(seconds, str): + return None + try: + return int(float(seconds)) + except ValueError: + return None + + +def _read_all_bytes(file_obj: object) -> bytes: + if isinstance(file_obj, (BytesIO, BufferedReader)): + current_position: Final = file_obj.tell() + file_obj.seek(0) + content: Final = file_obj.read() + file_obj.seek(current_position) + return content + if isinstance(file_obj, bytes): + return file_obj + if isinstance(file_obj, bytearray): + return bytes(file_obj) + read: Final = getattr(file_obj, "read", None) + if callable(read): + data: Final = read() + if isinstance(data, bytes): + return data + raise ValueError("input_reference must be a URL string, bytes, or a file object") + + +def _image_url(image: object) -> str: + if isinstance(image, str): + return image + content_type: Final = ImageEditRequestUtils.get_image_content_type(image) + encoded: Final = base64.b64encode(_read_all_bytes(image)).decode("utf-8") + return f"data:{content_type};base64,{encoded}" + + +def _media_reference_type(model: str) -> str | None: + """Models without the field predate the media array and take the image on the flat ``img_url`` field.""" + try: + info: Final = get_model_info(model=model, custom_llm_provider="dashscope") + except Exception: + return None + provider_specific: Final = info.get("provider_specific_entry") + value: Final = provider_specific.get(_REFERENCE_INPUT_FIELD) if isinstance(provider_specific, Mapping) else None + return value if isinstance(value, str) and value in _MEDIA_REFERENCE_TYPES else None + + +def _resolution_label(usage: _TaskUsage | None, requested_resolution: object) -> str | None: + if usage is not None and isinstance(usage.SR, int) and not isinstance(usage.SR, bool): + tier: Final = _RESOLUTION_TIERS.get(usage.SR) + if tier is not None: + return tier + if isinstance(requested_resolution, str) and requested_resolution.strip(): + return requested_resolution.strip().lower() + return None + + +def _video_usage(usage: _TaskUsage | None, requested: Mapping[str, object]) -> Mapping[str, object]: + """ + ``usage.duration`` is DashScope's billed duration, including input video seconds, so it beats the request. + """ + billed_duration: Final = usage.duration if usage is not None else None + requested_duration: Final = requested.get("duration") + duration_seconds: Final = ( + float(billed_duration) + if isinstance(billed_duration, (int, float)) and not isinstance(billed_duration, bool) + else float(requested_duration) + if isinstance(requested_duration, (int, float)) + and not isinstance(requested_duration, bool) + and requested_duration > 0 + else None + ) + resolution: Final = _resolution_label(usage, requested.get("resolution")) + return MappingProxyType( + { + key: value + for key, value in (("duration_seconds", duration_seconds), ("video_resolution", resolution)) + if value is not None + } + ) + + +def _timestamp(value: str | None) -> int | None: + """ + DashScope stamps times in UTC+8 with no offset. + """ + if not value: + return None + try: + parsed: Final = datetime.strptime(f"{value}+0800", "%Y-%m-%d %H:%M:%S.%f%z") + except ValueError: + return None + return int(parsed.timestamp()) + + +def _error_block(output: _TaskOutput) -> Mapping[str, object] | None: + if not (output.code or output.message): + return None + return MappingProxyType( + {key: value for key, value in (("code", output.code), ("message", output.message)) if value is not None} + ) + + +def _size_from_usage(usage: _TaskUsage | None) -> str | None: + if usage is None or not isinstance(usage.SR, int) or isinstance(usage.SR, bool) or usage.SR <= 0: + return None + if not usage.ratio or ":" not in usage.ratio: + return None + width_str, _, height_str = usage.ratio.partition(":") + if not (width_str.isdigit() and height_str.isdigit()): + return None + ratio_width: Final = int(width_str) + ratio_height: Final = int(height_str) + if not ratio_width or not ratio_height: + return None + shortest_ratio_side: Final = min(ratio_width, ratio_height) + return f"{usage.SR * ratio_width // shortest_ratio_side}x{usage.SR * ratio_height // shortest_ratio_side}" + + +def _video_object_from_task( + task: _TaskResponse, + model: str | None = None, + requested: Mapping[str, object] | None = None, +) -> VideoObject: + output: Final = task.output or _TaskOutput() + status: Final = _STATUS_MAP.get(output.task_status, "queued") + usage: Final = _video_usage(task.usage, requested or _EMPTY_PARAMS) + seconds: Final = usage.get("duration_seconds") + error_block: Final = _error_block(output) + return VideoObject( + id=output.task_id, + object="video", + status=status, + created_at=_timestamp(output.submit_time), + completed_at=_timestamp(output.end_time) if status == "completed" else None, + error=dict(error_block) if error_block is not None else None, # mutable-ok: VideoObject.error is a dict field + seconds=str(seconds) if seconds is not None else None, + size=_size_from_usage(task.usage), + model=model, + usage=dict(usage) if usage else None, # mutable-ok: VideoObject.usage is a dict field + ) + + +def _video_url_from_task(task: _TaskResponse) -> str: + output: Final = task.output or _TaskOutput() + if output.video_url: + return output.video_url + + if output.task_status in _PENDING_STATUSES: + raise ValueError(f"Video is still processing (status: {output.task_status}). Please wait and try again.") + if output.code or output.message: + raise ValueError(f"Video generation failed: {output.message or output.code}") + if output.task_status == "UNKNOWN": + raise ValueError("Task not found. DashScope task ids expire 24 hours after creation.") + raise ValueError("Video URL not found in task response. The task may not have succeeded yet.") + + +def _polled_model_id(logging_obj: object) -> str | None: + """ + The deployment the proxy routes by is the model encoded in the polled id, so it must survive into the returned id. + """ + litellm_params: Final = getattr(logging_obj, "litellm_params", None) + video_id: Final = litellm_params.get("video_id") if isinstance(litellm_params, Mapping) else None + if not isinstance(video_id, str): + return None + return decode_video_id_with_provider(video_id).get("model_id") or None + + +class DashScopeVideoConfig(BaseVideoConfig): + def get_supported_openai_params(self, model: str) -> list[str]: # mutable-ok: BaseVideoConfig contract returns list + return [ # mutable-ok: BaseVideoConfig contract returns list + "model", + "prompt", + "input_reference", + "seconds", + "size", + "user", + "extra_headers", + "media", + "resolution", + "ratio", + "duration", + "audio", + "seed", + "prompt_extend", + "watermark", + "negative_prompt", + "parameters", + *_LEGACY_INPUT_KEYS, + ] + + def map_openai_params( + self, + video_create_optional_params: VideoCreateOptionalRequestParams, + model: str, + drop_params: bool, + ) -> dict[str, object]: # mutable-ok: BaseVideoConfig contract + mapped_params: Final[dict[str, object]] = {} # mutable-ok: BaseVideoConfig contract; extra_body merges into it + for key, value in video_create_optional_params.items(): + if value is None or key in _DROP_FROM_REQUEST: + continue + if key == "seconds": + duration = _duration_param(value) + if duration is not None: + mapped_params["duration"] = duration + elif key == "size": + if not isinstance(value, str): + continue + ratio = _ratio_from_size(value) + if ratio is not None: + mapped_params.setdefault("ratio", ratio) + resolution = _resolution_from_size(value) + if resolution is not None: + mapped_params.setdefault("resolution", resolution) + elif key == "parameters": + try: + mapped_params.update(_PARAMETERS_ADAPTER.validate_python(value)) + except ValidationError as e: + raise ValueError("parameters must be an object of DashScope request fields") from e + else: + mapped_params[key] = value + return mapped_params + + def _resolve_api_key(self, api_key: str | None) -> str: + resolved_api_key: Final = api_key or get_secret_str("DASHSCOPE_API_KEY") + if resolved_api_key is None: + raise ValueError( + "DashScope API key is required. Set DASHSCOPE_API_KEY environment variable or pass api_key parameter." + ) + return resolved_api_key + + def _resolve_video_api_base(self, video_api_base: str | None) -> str: + return video_api_base or get_secret_str("DASHSCOPE_API_BASE_VIDEO") or DASHSCOPE_VIDEO_DEFAULT_API_BASE + + def validate_environment( + self, + headers: dict[str, str], # mutable-ok: BaseVideoConfig contract; handler expects a mutable headers dict + model: str, + api_key: str | None = None, + litellm_params: GenericLiteLLMParams | None = None, + ) -> dict[str, str]: # mutable-ok: BaseVideoConfig contract + resolved_api_key: Final = self._resolve_api_key( + api_key + or (litellm_params.api_key if litellm_params is not None and litellm_params.api_key else None) + or litellm.api_key + ) + + auth_headers: Final[dict[str, str]] = { # mutable-ok: httpx request headers are a mutable dict + "Authorization": f"Bearer {resolved_api_key}", + "Content-Type": "application/json", + "X-DashScope-Async": "enable", + } + headers.update(auth_headers) + return headers + + def get_complete_url( + self, + model: str, + api_base: str | None, + litellm_params: dict[str, object], # mutable-ok: BaseVideoConfig contract + ) -> str: + return _normalized_api_base(self._resolve_video_api_base(api_base)) + + def transform_video_create_request( + self, + model: str, + prompt: str, + api_base: str, + video_create_optional_request_params: dict[str, object], # mutable-ok: BaseVideoConfig contract + litellm_params: GenericLiteLLMParams, + headers: dict[str, str], # mutable-ok: BaseVideoConfig contract + ) -> tuple[dict[str, object], RequestFiles, str]: # mutable-ok: BaseVideoConfig contract + reference_field, reference_value = self._reference_input(model, video_create_optional_request_params) + passthrough_input: Final = tuple( + (key, video_create_optional_request_params[key]) + for key in _INPUT_KEYS + if video_create_optional_request_params.get(key) is not None + ) + reference_entry: Final = ( + ((reference_field, reference_value),) + if reference_field is not None and reference_field not in frozenset(key for key, _ in passthrough_input) + else () + ) + request_input: Final[dict[str, object]] = dict( # mutable-ok: request body dict, JSON-serialized by the handler + (("prompt", prompt), *passthrough_input, *reference_entry) + ) + + parameters: Final[dict[str, object]] = dict( # mutable-ok: request body dict, JSON-serialized by the handler + (key, video_create_optional_request_params[key]) + for key in _PARAMETER_KEYS + if video_create_optional_request_params.get(key) is not None + ) + if parameters.get("duration") == DASHSCOPE_SMART_DURATION: + raise UnsupportedParamsError( + message=( + "Smart duration (duration=-1) is not supported through litellm: the video is billed when the task " + "is created, before DashScope has picked its length. Pass an explicit duration in seconds." + ), + model=model, + llm_provider="dashscope", + ) + + request_data: Final[dict[str, object]] = { # mutable-ok: request body dict, JSON-serialized by the handler + "model": model, + "input": request_input, + } + if parameters: + request_data["parameters"] = parameters + + return request_data, (), f"{api_base}{DASHSCOPE_VIDEO_SYNTHESIS_PATH}" + + @staticmethod + def _reference_input( + model: str, + video_create_optional_request_params: Mapping[str, object], + ) -> tuple[str | None, object]: + input_reference: Final = video_create_optional_request_params.get("input_reference") + if input_reference is None: + return None, None + image_url: Final = _image_url(input_reference) + media_type: Final = _media_reference_type(model) + if media_type is None: + return "img_url", image_url + # a plain dict because json.dumps cannot serialize a MappingProxyType + return "media", (dict[str, object](type=media_type, url=image_url),) + + def transform_video_create_response( + self, + model: str, + raw_response: httpx.Response, + logging_obj: "Logging", + custom_llm_provider: str | None = None, + request_data: dict[str, object] | None = None, # mutable-ok: BaseVideoConfig contract + ) -> VideoObject: + task: Final = _parse_task_response(raw_response) + self._raise_for_task_error(task, raw_response) + requested: Final = self._requested_parameters(request_data) + video_obj: Final = _video_object_from_task(task, model=model, requested=requested) + if custom_llm_provider and video_obj.id: + video_obj.id = encode_video_id_with_provider(video_obj.id, custom_llm_provider, model) + return video_obj + + @staticmethod + def _requested_parameters(request_data: Mapping[str, object] | None) -> Mapping[str, object]: + """ + The create call is the only billed one, so an omitted duration or tier bills DashScope's defaults, not zero. + """ + parameters: Final = (request_data or _EMPTY_PARAMS).get("parameters") + requested: Final = ( + _PARAMETERS_ADAPTER.validate_python(parameters) if isinstance(parameters, Mapping) else _EMPTY_PARAMS + ) + return MappingProxyType( + { + "duration": DASHSCOPE_DEFAULT_DURATION_SECONDS, + "resolution": DASHSCOPE_DEFAULT_RESOLUTION, + **requested, + } + ) + + def _raise_for_task_error(self, task: _TaskResponse, raw_response: httpx.Response) -> None: + """ + Create and lookup both report failures as a 200 with top-level code and message and no output. + """ + if task.output is not None or not (task.code or task.message): + return + raise DashScopeVideoError( + status_code=raw_response.status_code, + message=task.message or task.code or "DashScope video request failed", + headers=raw_response.headers, + response=raw_response, + ) + + def transform_video_status_retrieve_request( + self, + video_id: str, + api_base: str, + litellm_params: GenericLiteLLMParams, + headers: dict[str, str], # mutable-ok: BaseVideoConfig contract + ) -> tuple[str, dict[str, object]]: # mutable-ok: BaseVideoConfig contract + return self._task_url(video_id, api_base), _EMPTY_PARAMS + + @staticmethod + def _task_url(video_id: str, api_base: str) -> str: + original_task_id: Final = extract_original_video_id(video_id) + encoded_task_id: Final = encode_url_path_segment(original_task_id, field_name="video_id") + return f"{api_base}{DASHSCOPE_TASKS_PATH}/{encoded_task_id}" + + def transform_video_status_retrieve_response( + self, + raw_response: httpx.Response, + logging_obj: "Logging", + custom_llm_provider: str | None = None, + client: "HTTPHandler | None" = None, + ) -> VideoObject: + task: Final = _parse_task_response(raw_response) + self._raise_for_task_error(task, raw_response) + model: Final = _polled_model_id(logging_obj) + video_obj: Final = _video_object_from_task(task, model=model) + if custom_llm_provider and video_obj.id: + video_obj.id = encode_video_id_with_provider(video_obj.id, custom_llm_provider, model) + return video_obj + + def transform_video_content_request( + self, + video_id: str, + api_base: str, + litellm_params: GenericLiteLLMParams, + headers: dict[str, str], # mutable-ok: BaseVideoConfig contract + variant: str | None = None, + ) -> tuple[str, dict[str, object]]: # mutable-ok: BaseVideoConfig contract + return self._task_url(video_id, api_base), _EMPTY_PARAMS + + def transform_video_content_response( + self, + raw_response: httpx.Response, + logging_obj: "Logging", + ) -> bytes: + video_url: Final = _video_url_from_task(_parse_task_response(raw_response)) + + httpx_client: Final = _get_httpx_client() + video_response: Final = httpx_client.get(video_url) # pyright: ignore[reportUnknownMemberType] # HTTPHandler.get stub leaves params/headers untyped + video_response.raise_for_status() + + return video_response.content + + async def async_transform_video_content_response( + self, + raw_response: httpx.Response, + logging_obj: "Logging", + ) -> bytes: + video_url: Final = _video_url_from_task(_parse_task_response(raw_response)) + + async_httpx_client: Final = get_async_httpx_client( + llm_provider=litellm.LlmProviders.DASHSCOPE, + ) + video_response: Final = await async_httpx_client.get(video_url) # pyright: ignore[reportUnknownMemberType] # HTTPHandler.get stub leaves params/headers untyped + video_response.raise_for_status() + + return video_response.content + + def transform_video_remix_request( + self, + video_id: str, + prompt: str, + api_base: str, + litellm_params: GenericLiteLLMParams, + headers: dict[str, str], # mutable-ok: BaseVideoConfig contract + extra_body: dict[str, object] | None = None, # mutable-ok: BaseVideoConfig contract + ) -> tuple[str, dict[str, object]]: # mutable-ok: BaseVideoConfig contract + raise NotImplementedError( + "Video remix is not supported by DashScope. Wan 3.0 edits and extends a video by passing it back as a " + "reference_video media entry on video_generation() with an editing or extension prompt." + ) + + def transform_video_remix_response( + self, + raw_response: httpx.Response, + logging_obj: "Logging", + custom_llm_provider: str | None = None, + ) -> VideoObject: + raise NotImplementedError("Video remix is not supported by DashScope.") + + def transform_video_list_request( + self, + api_base: str, + litellm_params: GenericLiteLLMParams, + headers: dict[str, str], # mutable-ok: BaseVideoConfig contract + after: str | None = None, + limit: int | None = None, + order: str | None = None, + extra_query: dict[str, object] | None = None, # mutable-ok: BaseVideoConfig contract + ) -> tuple[str, dict[str, object]]: # mutable-ok: BaseVideoConfig contract + raise NotImplementedError( + "Video list is not supported by DashScope. Retrieve tasks individually by task id within their 24 hour " + "retention window." + ) + + def transform_video_list_response( + self, + raw_response: httpx.Response, + logging_obj: "Logging", + custom_llm_provider: str | None = None, + ) -> dict[str, str]: # mutable-ok: BaseVideoConfig contract + raise NotImplementedError("Video list is not supported by DashScope.") + + def transform_video_delete_request( + self, + video_id: str, + api_base: str, + litellm_params: GenericLiteLLMParams, + headers: dict[str, str], # mutable-ok: BaseVideoConfig contract + ) -> tuple[str, dict[str, object]]: # mutable-ok: BaseVideoConfig contract + raise NotImplementedError( + "Video delete is not supported by DashScope. Tasks and their artifacts expire 24 hours after creation." + ) + + def transform_video_delete_response( + self, + raw_response: httpx.Response, + logging_obj: "Logging", + ) -> VideoObject: + raise NotImplementedError("Video delete is not supported by DashScope.") + + def get_error_class( + self, + error_message: str, + status_code: int, + headers: dict[str, str] | httpx.Headers, # mutable-ok: BaseVideoConfig contract + ) -> BaseLLMException: + return DashScopeVideoError( + status_code=status_code, + message=error_message, + headers=headers, + ) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index fc48c17b506..ffa523e8120 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -17019,6 +17019,281 @@ "/v1/images/generations" ] }, + "dashscope/wan3.0-video": { + "litellm_provider": "dashscope", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/wan3-0-video", + "output_cost_per_second": 0.165025, + "output_cost_per_second_480p": 0.041256, + "output_cost_per_second_720p": 0.082513, + "output_cost_per_second_1080p": 0.165025, + "supported_modalities": [ + "text", + "image", + "video", + "audio" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "Wan 3.0 all-in-one video generation (text, first/last frame, multimodal reference, edit, extend). Billed duration is usage.duration, which adds input video seconds for the edit and extend modes. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "dashscope/wan3.0-video-prime": { + "litellm_provider": "dashscope", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/wan3-0-video-prime", + "output_cost_per_second": 0.254399, + "output_cost_per_second_480p": 0.0636, + "output_cost_per_second_720p": 0.127199, + "output_cost_per_second_1080p": 0.254399, + "supported_modalities": [ + "text", + "image", + "video", + "audio" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "Wan 3.0 fast tier. Billed duration is usage.duration, which adds input video seconds for the edit and extend modes. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "dashscope/happyhorse-1.1-t2v": { + "litellm_provider": "dashscope", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-t2v", + "output_cost_per_second": 0.165026, + "output_cost_per_second_480p": 0.0618845, + "output_cost_per_second_720p": 0.123769, + "output_cost_per_second_1080p": 0.165026, + "supported_modalities": [ + "text" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "HappyHorse 1.1 text-to-video. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "dashscope/happyhorse-1.1-i2v": { + "litellm_provider": "dashscope", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-i2v", + "output_cost_per_second": 0.165026, + "output_cost_per_second_480p": 0.0618845, + "output_cost_per_second_720p": 0.123769, + "output_cost_per_second_1080p": 0.165026, + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "HappyHorse 1.1 image-to-video from a first frame. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "dashscope/happyhorse-1.1-r2v": { + "litellm_provider": "dashscope", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-r2v", + "output_cost_per_second": 0.165026, + "output_cost_per_second_480p": 0.0618845, + "output_cost_per_second_720p": 0.123769, + "output_cost_per_second_1080p": 0.165026, + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "reference_image" + }, + "metadata": { + "comment": "HappyHorse 1.1 reference-to-video, 1 to 9 reference images. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "dashscope/happyhorse-1.0-t2v": { + "litellm_provider": "dashscope", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-t2v", + "output_cost_per_second": 0.220034, + "output_cost_per_second_720p": 0.123769, + "output_cost_per_second_1080p": 0.220034, + "supported_modalities": [ + "text" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "HappyHorse 1.0 text-to-video. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "dashscope/happyhorse-1.0-i2v": { + "litellm_provider": "dashscope", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-i2v", + "output_cost_per_second": 0.220034, + "output_cost_per_second_720p": 0.123769, + "output_cost_per_second_1080p": 0.220034, + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "HappyHorse 1.0 image-to-video from a first frame. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "dashscope/happyhorse-1.0-r2v": { + "litellm_provider": "dashscope", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-r2v", + "output_cost_per_second": 0.220034, + "output_cost_per_second_720p": 0.123769, + "output_cost_per_second_1080p": 0.220034, + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "reference_image" + }, + "metadata": { + "comment": "HappyHorse 1.0 reference-to-video. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "dashscope/wan2.7-t2v": { + "litellm_provider": "dashscope", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-t2v", + "output_cost_per_second": 0.143353, + "output_cost_per_second_720p": 0.086012, + "output_cost_per_second_1080p": 0.143353, + "supported_modalities": [ + "text", + "audio" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "Wan 2.7 text-to-video. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "dashscope/wan2.7-i2v": { + "litellm_provider": "dashscope", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-i2v", + "output_cost_per_second": 0.143353, + "output_cost_per_second_720p": 0.086012, + "output_cost_per_second_1080p": 0.143353, + "supported_modalities": [ + "text", + "image", + "video", + "audio" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "Wan 2.7 image-to-video (first frame, first and last frame, continuation). No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "dashscope/wan2.7-r2v": { + "litellm_provider": "dashscope", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-r2v", + "output_cost_per_second": 0.143353, + "output_cost_per_second_720p": 0.086012, + "output_cost_per_second_1080p": 0.143353, + "supported_modalities": [ + "text", + "image", + "video", + "audio" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "reference_image" + }, + "metadata": { + "comment": "Wan 2.7 reference-to-video. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, "qwencloud/deepseek-v4-flash": { "cache_read_input_token_cost": 4e-08, "input_cost_per_token": 2e-07, @@ -17971,6 +18246,281 @@ "/v1/images/generations" ] }, + "qwencloud/wan3.0-video": { + "litellm_provider": "qwencloud", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/wan3-0-video", + "output_cost_per_second": 0.2, + "output_cost_per_second_480p": 0.05, + "output_cost_per_second_720p": 0.1, + "output_cost_per_second_1080p": 0.2, + "supported_modalities": [ + "text", + "image", + "video", + "audio" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "Wan 3.0 all-in-one video generation (text, first/last frame, multimodal reference, edit, extend). Billed duration is usage.duration, which adds input video seconds for the edit and extend modes. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwencloud/wan3.0-video-prime": { + "litellm_provider": "qwencloud", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/wan3-0-video-prime", + "output_cost_per_second": 0.28, + "output_cost_per_second_480p": 0.068, + "output_cost_per_second_720p": 0.14, + "output_cost_per_second_1080p": 0.28, + "supported_modalities": [ + "text", + "image", + "video", + "audio" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "Wan 3.0 fast tier. Billed duration is usage.duration, which adds input video seconds for the edit and extend modes. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwencloud/happyhorse-1.1-t2v": { + "litellm_provider": "qwencloud", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-t2v", + "output_cost_per_second": 0.18, + "output_cost_per_second_480p": 0.07, + "output_cost_per_second_720p": 0.14, + "output_cost_per_second_1080p": 0.18, + "supported_modalities": [ + "text" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "HappyHorse 1.1 text-to-video. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwencloud/happyhorse-1.1-i2v": { + "litellm_provider": "qwencloud", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-i2v", + "output_cost_per_second": 0.18, + "output_cost_per_second_480p": 0.07, + "output_cost_per_second_720p": 0.14, + "output_cost_per_second_1080p": 0.18, + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "HappyHorse 1.1 image-to-video from a first frame. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwencloud/happyhorse-1.1-r2v": { + "litellm_provider": "qwencloud", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-r2v", + "output_cost_per_second": 0.18, + "output_cost_per_second_480p": 0.07, + "output_cost_per_second_720p": 0.14, + "output_cost_per_second_1080p": 0.18, + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "reference_image" + }, + "metadata": { + "comment": "HappyHorse 1.1 reference-to-video, 1 to 9 reference images. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwencloud/happyhorse-1.0-t2v": { + "litellm_provider": "qwencloud", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-t2v", + "output_cost_per_second": 0.24, + "output_cost_per_second_720p": 0.14, + "output_cost_per_second_1080p": 0.24, + "supported_modalities": [ + "text" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "HappyHorse 1.0 text-to-video. No 480P tier. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwencloud/happyhorse-1.0-i2v": { + "litellm_provider": "qwencloud", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-i2v", + "output_cost_per_second": 0.24, + "output_cost_per_second_720p": 0.14, + "output_cost_per_second_1080p": 0.24, + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "HappyHorse 1.0 image-to-video from a first frame. No 480P tier. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwencloud/happyhorse-1.0-r2v": { + "litellm_provider": "qwencloud", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-r2v", + "output_cost_per_second": 0.24, + "output_cost_per_second_720p": 0.14, + "output_cost_per_second_1080p": 0.24, + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "reference_image" + }, + "metadata": { + "comment": "HappyHorse 1.0 reference-to-video. No 480P tier. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwencloud/wan2.7-t2v": { + "litellm_provider": "qwencloud", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-t2v", + "output_cost_per_second": 0.15, + "output_cost_per_second_720p": 0.1, + "output_cost_per_second_1080p": 0.15, + "supported_modalities": [ + "text", + "audio" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "Wan 2.7 text-to-video. No 480P tier. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwencloud/wan2.7-i2v": { + "litellm_provider": "qwencloud", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-i2v", + "output_cost_per_second": 0.15, + "output_cost_per_second_720p": 0.1, + "output_cost_per_second_1080p": 0.15, + "supported_modalities": [ + "text", + "image", + "video", + "audio" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "Wan 2.7 image-to-video (first frame, first and last frame, continuation). No 480P tier. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwencloud/wan2.7-r2v": { + "litellm_provider": "qwencloud", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-r2v", + "output_cost_per_second": 0.15, + "output_cost_per_second_720p": 0.1, + "output_cost_per_second_1080p": 0.15, + "supported_modalities": [ + "text", + "image", + "video", + "audio" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "reference_image" + }, + "metadata": { + "comment": "Wan 2.7 reference-to-video. No 480P tier. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, "qwen_ai_platform/deepseek-v4-flash": { "cache_read_input_token_cost": 4e-08, "input_cost_per_token": 2e-07, @@ -18963,6 +19513,281 @@ "/v1/images/generations" ] }, + "qwen_ai_platform/wan3.0-video": { + "litellm_provider": "qwen_ai_platform", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/wan3-0-video", + "output_cost_per_second": 0.165025, + "output_cost_per_second_480p": 0.041256, + "output_cost_per_second_720p": 0.082513, + "output_cost_per_second_1080p": 0.165025, + "supported_modalities": [ + "text", + "image", + "video", + "audio" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "Wan 3.0 all-in-one video generation (text, first/last frame, multimodal reference, edit, extend). Billed duration is usage.duration, which adds input video seconds for the edit and extend modes. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwen_ai_platform/wan3.0-video-prime": { + "litellm_provider": "qwen_ai_platform", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/wan3-0-video-prime", + "output_cost_per_second": 0.254399, + "output_cost_per_second_480p": 0.0636, + "output_cost_per_second_720p": 0.127199, + "output_cost_per_second_1080p": 0.254399, + "supported_modalities": [ + "text", + "image", + "video", + "audio" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "Wan 3.0 fast tier. Billed duration is usage.duration, which adds input video seconds for the edit and extend modes. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwen_ai_platform/happyhorse-1.1-t2v": { + "litellm_provider": "qwen_ai_platform", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-t2v", + "output_cost_per_second": 0.165026, + "output_cost_per_second_480p": 0.0618845, + "output_cost_per_second_720p": 0.123769, + "output_cost_per_second_1080p": 0.165026, + "supported_modalities": [ + "text" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "HappyHorse 1.1 text-to-video. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwen_ai_platform/happyhorse-1.1-i2v": { + "litellm_provider": "qwen_ai_platform", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-i2v", + "output_cost_per_second": 0.165026, + "output_cost_per_second_480p": 0.0618845, + "output_cost_per_second_720p": 0.123769, + "output_cost_per_second_1080p": 0.165026, + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "HappyHorse 1.1 image-to-video from a first frame. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwen_ai_platform/happyhorse-1.1-r2v": { + "litellm_provider": "qwen_ai_platform", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-r2v", + "output_cost_per_second": 0.165026, + "output_cost_per_second_480p": 0.0618845, + "output_cost_per_second_720p": 0.123769, + "output_cost_per_second_1080p": 0.165026, + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "reference_image" + }, + "metadata": { + "comment": "HappyHorse 1.1 reference-to-video, 1 to 9 reference images. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwen_ai_platform/happyhorse-1.0-t2v": { + "litellm_provider": "qwen_ai_platform", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-t2v", + "output_cost_per_second": 0.220034, + "output_cost_per_second_720p": 0.123769, + "output_cost_per_second_1080p": 0.220034, + "supported_modalities": [ + "text" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "HappyHorse 1.0 text-to-video. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwen_ai_platform/happyhorse-1.0-i2v": { + "litellm_provider": "qwen_ai_platform", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-i2v", + "output_cost_per_second": 0.220034, + "output_cost_per_second_720p": 0.123769, + "output_cost_per_second_1080p": 0.220034, + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "HappyHorse 1.0 image-to-video from a first frame. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwen_ai_platform/happyhorse-1.0-r2v": { + "litellm_provider": "qwen_ai_platform", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-r2v", + "output_cost_per_second": 0.220034, + "output_cost_per_second_720p": 0.123769, + "output_cost_per_second_1080p": 0.220034, + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "reference_image" + }, + "metadata": { + "comment": "HappyHorse 1.0 reference-to-video. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwen_ai_platform/wan2.7-t2v": { + "litellm_provider": "qwen_ai_platform", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-t2v", + "output_cost_per_second": 0.143353, + "output_cost_per_second_720p": 0.086012, + "output_cost_per_second_1080p": 0.143353, + "supported_modalities": [ + "text", + "audio" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "Wan 2.7 text-to-video. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwen_ai_platform/wan2.7-i2v": { + "litellm_provider": "qwen_ai_platform", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-i2v", + "output_cost_per_second": 0.143353, + "output_cost_per_second_720p": 0.086012, + "output_cost_per_second_1080p": 0.143353, + "supported_modalities": [ + "text", + "image", + "video", + "audio" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "Wan 2.7 image-to-video (first frame, first and last frame, continuation). No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwen_ai_platform/wan2.7-r2v": { + "litellm_provider": "qwen_ai_platform", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-r2v", + "output_cost_per_second": 0.143353, + "output_cost_per_second_720p": 0.086012, + "output_cost_per_second_1080p": 0.143353, + "supported_modalities": [ + "text", + "image", + "video", + "audio" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "reference_image" + }, + "metadata": { + "comment": "Wan 2.7 reference-to-video. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, "databricks/databricks-bge-large-en": { "cache_creation_input_token_cost": 1.0003e-07, "cache_read_input_token_cost": 1.0003e-07, diff --git a/litellm/utils.py b/litellm/utils.py index 13a46840431..efcc0a73f70 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -9626,6 +9626,16 @@ class ProviderConfigManager: from litellm.llms.vertex_ai.videos.transformation import VertexAIVideoConfig return VertexAIVideoConfig() + elif provider in ( + LlmProviders.DASHSCOPE, + LlmProviders.QWENCLOUD, + LlmProviders.QWEN_AI_PLATFORM, + ): + from litellm.llms.dashscope.common_utils import ( + get_dashscope_family_video_config, + ) + + return get_dashscope_family_video_config(provider.value) elif LlmProviders.RUNWAYML == provider: from litellm.llms.runwayml.videos.transformation import RunwayMLVideoConfig diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index fc48c17b506..ffa523e8120 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -17019,6 +17019,281 @@ "/v1/images/generations" ] }, + "dashscope/wan3.0-video": { + "litellm_provider": "dashscope", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/wan3-0-video", + "output_cost_per_second": 0.165025, + "output_cost_per_second_480p": 0.041256, + "output_cost_per_second_720p": 0.082513, + "output_cost_per_second_1080p": 0.165025, + "supported_modalities": [ + "text", + "image", + "video", + "audio" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "Wan 3.0 all-in-one video generation (text, first/last frame, multimodal reference, edit, extend). Billed duration is usage.duration, which adds input video seconds for the edit and extend modes. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "dashscope/wan3.0-video-prime": { + "litellm_provider": "dashscope", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/wan3-0-video-prime", + "output_cost_per_second": 0.254399, + "output_cost_per_second_480p": 0.0636, + "output_cost_per_second_720p": 0.127199, + "output_cost_per_second_1080p": 0.254399, + "supported_modalities": [ + "text", + "image", + "video", + "audio" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "Wan 3.0 fast tier. Billed duration is usage.duration, which adds input video seconds for the edit and extend modes. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "dashscope/happyhorse-1.1-t2v": { + "litellm_provider": "dashscope", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-t2v", + "output_cost_per_second": 0.165026, + "output_cost_per_second_480p": 0.0618845, + "output_cost_per_second_720p": 0.123769, + "output_cost_per_second_1080p": 0.165026, + "supported_modalities": [ + "text" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "HappyHorse 1.1 text-to-video. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "dashscope/happyhorse-1.1-i2v": { + "litellm_provider": "dashscope", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-i2v", + "output_cost_per_second": 0.165026, + "output_cost_per_second_480p": 0.0618845, + "output_cost_per_second_720p": 0.123769, + "output_cost_per_second_1080p": 0.165026, + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "HappyHorse 1.1 image-to-video from a first frame. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "dashscope/happyhorse-1.1-r2v": { + "litellm_provider": "dashscope", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-r2v", + "output_cost_per_second": 0.165026, + "output_cost_per_second_480p": 0.0618845, + "output_cost_per_second_720p": 0.123769, + "output_cost_per_second_1080p": 0.165026, + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "reference_image" + }, + "metadata": { + "comment": "HappyHorse 1.1 reference-to-video, 1 to 9 reference images. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "dashscope/happyhorse-1.0-t2v": { + "litellm_provider": "dashscope", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-t2v", + "output_cost_per_second": 0.220034, + "output_cost_per_second_720p": 0.123769, + "output_cost_per_second_1080p": 0.220034, + "supported_modalities": [ + "text" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "HappyHorse 1.0 text-to-video. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "dashscope/happyhorse-1.0-i2v": { + "litellm_provider": "dashscope", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-i2v", + "output_cost_per_second": 0.220034, + "output_cost_per_second_720p": 0.123769, + "output_cost_per_second_1080p": 0.220034, + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "HappyHorse 1.0 image-to-video from a first frame. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "dashscope/happyhorse-1.0-r2v": { + "litellm_provider": "dashscope", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-r2v", + "output_cost_per_second": 0.220034, + "output_cost_per_second_720p": 0.123769, + "output_cost_per_second_1080p": 0.220034, + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "reference_image" + }, + "metadata": { + "comment": "HappyHorse 1.0 reference-to-video. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "dashscope/wan2.7-t2v": { + "litellm_provider": "dashscope", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-t2v", + "output_cost_per_second": 0.143353, + "output_cost_per_second_720p": 0.086012, + "output_cost_per_second_1080p": 0.143353, + "supported_modalities": [ + "text", + "audio" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "Wan 2.7 text-to-video. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "dashscope/wan2.7-i2v": { + "litellm_provider": "dashscope", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-i2v", + "output_cost_per_second": 0.143353, + "output_cost_per_second_720p": 0.086012, + "output_cost_per_second_1080p": 0.143353, + "supported_modalities": [ + "text", + "image", + "video", + "audio" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "Wan 2.7 image-to-video (first frame, first and last frame, continuation). No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "dashscope/wan2.7-r2v": { + "litellm_provider": "dashscope", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-r2v", + "output_cost_per_second": 0.143353, + "output_cost_per_second_720p": 0.086012, + "output_cost_per_second_1080p": 0.143353, + "supported_modalities": [ + "text", + "image", + "video", + "audio" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "reference_image" + }, + "metadata": { + "comment": "Wan 2.7 reference-to-video. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, "qwencloud/deepseek-v4-flash": { "cache_read_input_token_cost": 4e-08, "input_cost_per_token": 2e-07, @@ -17971,6 +18246,281 @@ "/v1/images/generations" ] }, + "qwencloud/wan3.0-video": { + "litellm_provider": "qwencloud", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/wan3-0-video", + "output_cost_per_second": 0.2, + "output_cost_per_second_480p": 0.05, + "output_cost_per_second_720p": 0.1, + "output_cost_per_second_1080p": 0.2, + "supported_modalities": [ + "text", + "image", + "video", + "audio" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "Wan 3.0 all-in-one video generation (text, first/last frame, multimodal reference, edit, extend). Billed duration is usage.duration, which adds input video seconds for the edit and extend modes. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwencloud/wan3.0-video-prime": { + "litellm_provider": "qwencloud", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/wan3-0-video-prime", + "output_cost_per_second": 0.28, + "output_cost_per_second_480p": 0.068, + "output_cost_per_second_720p": 0.14, + "output_cost_per_second_1080p": 0.28, + "supported_modalities": [ + "text", + "image", + "video", + "audio" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "Wan 3.0 fast tier. Billed duration is usage.duration, which adds input video seconds for the edit and extend modes. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwencloud/happyhorse-1.1-t2v": { + "litellm_provider": "qwencloud", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-t2v", + "output_cost_per_second": 0.18, + "output_cost_per_second_480p": 0.07, + "output_cost_per_second_720p": 0.14, + "output_cost_per_second_1080p": 0.18, + "supported_modalities": [ + "text" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "HappyHorse 1.1 text-to-video. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwencloud/happyhorse-1.1-i2v": { + "litellm_provider": "qwencloud", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-i2v", + "output_cost_per_second": 0.18, + "output_cost_per_second_480p": 0.07, + "output_cost_per_second_720p": 0.14, + "output_cost_per_second_1080p": 0.18, + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "HappyHorse 1.1 image-to-video from a first frame. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwencloud/happyhorse-1.1-r2v": { + "litellm_provider": "qwencloud", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-r2v", + "output_cost_per_second": 0.18, + "output_cost_per_second_480p": 0.07, + "output_cost_per_second_720p": 0.14, + "output_cost_per_second_1080p": 0.18, + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "reference_image" + }, + "metadata": { + "comment": "HappyHorse 1.1 reference-to-video, 1 to 9 reference images. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwencloud/happyhorse-1.0-t2v": { + "litellm_provider": "qwencloud", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-t2v", + "output_cost_per_second": 0.24, + "output_cost_per_second_720p": 0.14, + "output_cost_per_second_1080p": 0.24, + "supported_modalities": [ + "text" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "HappyHorse 1.0 text-to-video. No 480P tier. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwencloud/happyhorse-1.0-i2v": { + "litellm_provider": "qwencloud", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-i2v", + "output_cost_per_second": 0.24, + "output_cost_per_second_720p": 0.14, + "output_cost_per_second_1080p": 0.24, + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "HappyHorse 1.0 image-to-video from a first frame. No 480P tier. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwencloud/happyhorse-1.0-r2v": { + "litellm_provider": "qwencloud", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-r2v", + "output_cost_per_second": 0.24, + "output_cost_per_second_720p": 0.14, + "output_cost_per_second_1080p": 0.24, + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "reference_image" + }, + "metadata": { + "comment": "HappyHorse 1.0 reference-to-video. No 480P tier. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwencloud/wan2.7-t2v": { + "litellm_provider": "qwencloud", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-t2v", + "output_cost_per_second": 0.15, + "output_cost_per_second_720p": 0.1, + "output_cost_per_second_1080p": 0.15, + "supported_modalities": [ + "text", + "audio" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "Wan 2.7 text-to-video. No 480P tier. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwencloud/wan2.7-i2v": { + "litellm_provider": "qwencloud", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-i2v", + "output_cost_per_second": 0.15, + "output_cost_per_second_720p": 0.1, + "output_cost_per_second_1080p": 0.15, + "supported_modalities": [ + "text", + "image", + "video", + "audio" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "Wan 2.7 image-to-video (first frame, first and last frame, continuation). No 480P tier. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwencloud/wan2.7-r2v": { + "litellm_provider": "qwencloud", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-r2v", + "output_cost_per_second": 0.15, + "output_cost_per_second_720p": 0.1, + "output_cost_per_second_1080p": 0.15, + "supported_modalities": [ + "text", + "image", + "video", + "audio" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "reference_image" + }, + "metadata": { + "comment": "Wan 2.7 reference-to-video. No 480P tier. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, "qwen_ai_platform/deepseek-v4-flash": { "cache_read_input_token_cost": 4e-08, "input_cost_per_token": 2e-07, @@ -18963,6 +19513,281 @@ "/v1/images/generations" ] }, + "qwen_ai_platform/wan3.0-video": { + "litellm_provider": "qwen_ai_platform", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/wan3-0-video", + "output_cost_per_second": 0.165025, + "output_cost_per_second_480p": 0.041256, + "output_cost_per_second_720p": 0.082513, + "output_cost_per_second_1080p": 0.165025, + "supported_modalities": [ + "text", + "image", + "video", + "audio" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "Wan 3.0 all-in-one video generation (text, first/last frame, multimodal reference, edit, extend). Billed duration is usage.duration, which adds input video seconds for the edit and extend modes. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwen_ai_platform/wan3.0-video-prime": { + "litellm_provider": "qwen_ai_platform", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/wan3-0-video-prime", + "output_cost_per_second": 0.254399, + "output_cost_per_second_480p": 0.0636, + "output_cost_per_second_720p": 0.127199, + "output_cost_per_second_1080p": 0.254399, + "supported_modalities": [ + "text", + "image", + "video", + "audio" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "Wan 3.0 fast tier. Billed duration is usage.duration, which adds input video seconds for the edit and extend modes. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwen_ai_platform/happyhorse-1.1-t2v": { + "litellm_provider": "qwen_ai_platform", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-t2v", + "output_cost_per_second": 0.165026, + "output_cost_per_second_480p": 0.0618845, + "output_cost_per_second_720p": 0.123769, + "output_cost_per_second_1080p": 0.165026, + "supported_modalities": [ + "text" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "HappyHorse 1.1 text-to-video. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwen_ai_platform/happyhorse-1.1-i2v": { + "litellm_provider": "qwen_ai_platform", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-i2v", + "output_cost_per_second": 0.165026, + "output_cost_per_second_480p": 0.0618845, + "output_cost_per_second_720p": 0.123769, + "output_cost_per_second_1080p": 0.165026, + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "HappyHorse 1.1 image-to-video from a first frame. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwen_ai_platform/happyhorse-1.1-r2v": { + "litellm_provider": "qwen_ai_platform", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-r2v", + "output_cost_per_second": 0.165026, + "output_cost_per_second_480p": 0.0618845, + "output_cost_per_second_720p": 0.123769, + "output_cost_per_second_1080p": 0.165026, + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "reference_image" + }, + "metadata": { + "comment": "HappyHorse 1.1 reference-to-video, 1 to 9 reference images. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwen_ai_platform/happyhorse-1.0-t2v": { + "litellm_provider": "qwen_ai_platform", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-t2v", + "output_cost_per_second": 0.220034, + "output_cost_per_second_720p": 0.123769, + "output_cost_per_second_1080p": 0.220034, + "supported_modalities": [ + "text" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "HappyHorse 1.0 text-to-video. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwen_ai_platform/happyhorse-1.0-i2v": { + "litellm_provider": "qwen_ai_platform", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-i2v", + "output_cost_per_second": 0.220034, + "output_cost_per_second_720p": 0.123769, + "output_cost_per_second_1080p": 0.220034, + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "HappyHorse 1.0 image-to-video from a first frame. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwen_ai_platform/happyhorse-1.0-r2v": { + "litellm_provider": "qwen_ai_platform", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-r2v", + "output_cost_per_second": 0.220034, + "output_cost_per_second_720p": 0.123769, + "output_cost_per_second_1080p": 0.220034, + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "reference_image" + }, + "metadata": { + "comment": "HappyHorse 1.0 reference-to-video. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwen_ai_platform/wan2.7-t2v": { + "litellm_provider": "qwen_ai_platform", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-t2v", + "output_cost_per_second": 0.143353, + "output_cost_per_second_720p": 0.086012, + "output_cost_per_second_1080p": 0.143353, + "supported_modalities": [ + "text", + "audio" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "Wan 2.7 text-to-video. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwen_ai_platform/wan2.7-i2v": { + "litellm_provider": "qwen_ai_platform", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-i2v", + "output_cost_per_second": 0.143353, + "output_cost_per_second_720p": 0.086012, + "output_cost_per_second_1080p": 0.143353, + "supported_modalities": [ + "text", + "image", + "video", + "audio" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "first_frame" + }, + "metadata": { + "comment": "Wan 2.7 image-to-video (first frame, first and last frame, continuation). No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, + "qwen_ai_platform/wan2.7-r2v": { + "litellm_provider": "qwen_ai_platform", + "mode": "video_generation", + "source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-r2v", + "output_cost_per_second": 0.143353, + "output_cost_per_second_720p": 0.086012, + "output_cost_per_second_1080p": 0.143353, + "supported_modalities": [ + "text", + "image", + "video", + "audio" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "provider_specific_entry": { + "dashscope_video_reference_input": "reference_image" + }, + "metadata": { + "comment": "Wan 2.7 reference-to-video. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier." + } + }, "databricks/databricks-bge-large-en": { "cache_creation_input_token_cost": 1.0003e-07, "cache_read_input_token_cost": 1.0003e-07, diff --git a/tests/unit/llms/dashscope/test_dashscope_video_transformation.py b/tests/unit/llms/dashscope/test_dashscope_video_transformation.py new file mode 100644 index 00000000000..2f37f5ebe25 --- /dev/null +++ b/tests/unit/llms/dashscope/test_dashscope_video_transformation.py @@ -0,0 +1,818 @@ +""" +Tests for DashScope (Wan 3.0 / Wan 2.7 / HappyHorse) video generation transformation. +""" + +import base64 +import io +import json +from datetime import datetime, timedelta, timezone +from typing import Final +from unittest.mock import Mock + +import httpx +import pytest + +from litellm.exceptions import UnsupportedParamsError +from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler +from litellm.llms.dashscope.qwen_ai_platform import QwenAIPlatformVideoConfig +from litellm.llms.dashscope.qwencloud import QwenCloudVideoConfig +from litellm.llms.dashscope.videos.transformation import ( + DashScopeVideoConfig, + DashScopeVideoError, + _parse_task_response, + _video_url_from_task, +) +from litellm.types.router import GenericLiteLLMParams +from litellm.types.videos.utils import ( + decode_video_id_with_provider, + encode_video_id_with_provider, +) +from litellm.videos.main import avideo_generation + +PNG_BYTES = base64.b64decode( + "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg==" +) + +API_BASE = "https://dashscope.aliyuncs.com" +SYNTHESIS_URL = f"{API_BASE}/api/v1/services/aigc/video-generation/video-synthesis" + + +def _mock_response(payload: dict, status_code: int = 200) -> Mock: + mock_response = Mock(spec=httpx.Response) + mock_response.json.return_value = payload + mock_response.status_code = status_code + mock_response.headers = httpx.Headers() + return mock_response + + +def _create(params: dict, model: str = "wan3.0-video", prompt: str = "a cat on a roof"): + return DashScopeVideoConfig().transform_video_create_request( + model=model, + prompt=prompt, + api_base=API_BASE, + video_create_optional_request_params=params, + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + +class TestDashScopeVideoCreateRequest: + def test_body_is_nested_into_input_and_parameters(self): + """DashScope rejects a flat Sora-shaped body: prompt and media belong + under ``input``, everything else under ``parameters``.""" + data, files, url = _create({"resolution": "720P", "duration": 10, "ratio": "16:9"}) + + assert url == SYNTHESIS_URL + assert files == () + assert data == { + "model": "wan3.0-video", + "input": {"prompt": "a cat on a roof"}, + "parameters": {"resolution": "720P", "ratio": "16:9", "duration": 10}, + } + + def test_prompt_only_request_omits_parameters_block(self): + """With nothing to configure, DashScope's documented text-to-video body + is just model + input, so no empty parameters block is sent.""" + data, _, _ = _create({}) + + assert data == {"model": "wan3.0-video", "input": {"prompt": "a cat on a roof"}} + + def test_media_array_passes_through_into_input(self): + media = [ + {"type": "reference_image", "url": "https://x/1.png"}, + {"type": "reference_audio", "url": "https://x/1.mp3"}, + ] + data, _, _ = _create({"media": media, "duration": 5}) + + assert data["input"]["media"] == media + assert "media" not in data["parameters"] + + def test_negative_prompt_goes_to_input_not_parameters(self): + data, _, _ = _create({"negative_prompt": "blurry", "seed": 7}) + + assert data["input"]["negative_prompt"] == "blurry" + assert data["parameters"] == {"seed": 7} + + +class TestDashScopeVideoInputReference: + def test_wan3_file_reference_becomes_first_frame_media_data_uri(self): + """Wan 3.0 takes references as typed media entries; a file must arrive + base64-encoded as a data URI rather than as an unusable file object.""" + data, _, _ = _create({"input_reference": io.BytesIO(PNG_BYTES)}) + + media = data["input"]["media"] + assert len(media) == 1 + assert media[0]["type"] == "first_frame" + assert media[0]["url"].startswith("data:image/png;base64,") + assert base64.b64decode(media[0]["url"].split(",", 1)[1]) == PNG_BYTES + + def test_wan3_url_reference_is_passed_through_unencoded(self): + data, _, _ = _create({"input_reference": "https://x/first.png"}) + + assert data["input"]["media"] == ({"type": "first_frame", "url": "https://x/first.png"},) + + @pytest.mark.parametrize( + "model", ["wan2.7-i2v", "wan2.7-i2v-2026-04-25", "happyhorse-1.1-i2v", "happyhorse-1.0-i2v"] + ) + def test_current_i2v_models_take_the_reference_as_a_first_frame_media_entry(self, model): + """Wan 2.7 and HappyHorse image-to-video take the typed media array; + a flat img_url is only understood by the legacy Wan 2.6 models.""" + data, _, _ = _create({"input_reference": "https://x/first.png"}, model=model) + + assert data["input"]["media"] == ({"type": "first_frame", "url": "https://x/first.png"},) + assert "img_url" not in data["input"] + + @pytest.mark.parametrize("model", ["happyhorse-1.1-r2v", "happyhorse-1.0-r2v", "wan2.7-r2v"]) + def test_reference_to_video_models_take_the_reference_as_a_reference_image(self, model): + """The r2v models reject first_frame and only accept reference_image.""" + data, _, _ = _create({"input_reference": "https://x/subject.png"}, model=model) + + assert data["input"]["media"] == ({"type": "reference_image", "url": "https://x/subject.png"},) + + @pytest.mark.parametrize( + "model", ["wan2.6-i2v-flash", "wan2.5-i2v-preview", "wan2.2-i2v-plus", "wanx2.1-i2v-turbo"] + ) + def test_legacy_models_take_the_reference_on_the_flat_img_url_field(self, model): + """Wan 2.6 and earlier predate the media array and would reject it.""" + data, _, _ = _create({"input_reference": "https://x/first.png"}, model=model) + + assert data["input"]["img_url"] == "https://x/first.png" + assert "media" not in data["input"] + + def test_explicit_media_array_wins_over_input_reference(self): + """Only the full media array can express multi-role references, so a + caller-supplied one must not be clobbered by input_reference.""" + media = [{"type": "reference_video", "url": "https://x/v.mp4"}] + data, _, _ = _create({"media": media, "input_reference": "https://x/first.png"}) + + assert data["input"]["media"] == media + + def test_legacy_explicit_img_url_wins_over_input_reference(self): + data, _, _ = _create( + {"img_url": "https://x/explicit.png", "input_reference": "https://x/other.png"}, + model="wan2.6-i2v-flash", + ) + + assert data["input"]["img_url"] == "https://x/explicit.png" + + +class TestDashScopeVideoMapOpenAIParams: + def test_seconds_becomes_duration(self): + mapped = DashScopeVideoConfig().map_openai_params( + video_create_optional_params={"seconds": "10"}, model="wan3.0-video", drop_params=False + ) + + assert mapped == {"duration": 10} + assert "seconds" not in mapped + + def test_size_becomes_both_ratio_and_resolution_tier(self): + """DashScope takes a ratio plus a named resolution tier, and bills per + second by that tier, so a pixel size must produce both or the request + silently falls back to the pricier 1080P default.""" + mapped = DashScopeVideoConfig().map_openai_params( + video_create_optional_params={"size": "1280x720"}, model="wan3.0-video", drop_params=False + ) + + assert mapped == {"ratio": "16:9", "resolution": "720P"} + + @pytest.mark.parametrize( + "size,expected_resolution", + [ + ("854x480", "480P"), + ("1280x720", "720P"), + ("1920x1080", "1080P"), + ("1080x1920", "1080P"), + ("3840x2160", "1080P"), + ], + ) + def test_resolution_tier_is_picked_from_the_shortest_side(self, size, expected_resolution): + mapped = DashScopeVideoConfig().map_openai_params( + video_create_optional_params={"size": size}, model="wan3.0-video", drop_params=False + ) + + assert mapped["resolution"] == expected_resolution + + def test_explicit_resolution_and_ratio_win_over_size(self): + mapped = DashScopeVideoConfig().map_openai_params( + video_create_optional_params={"size": "1280x720", "ratio": "4:3", "resolution": "1080P"}, + model="wan3.0-video", + drop_params=False, + ) + + assert mapped["ratio"] == "4:3" + assert mapped["resolution"] == "1080P" + + def test_openai_only_fields_are_dropped(self): + mapped = DashScopeVideoConfig().map_openai_params( + video_create_optional_params={"user": "u1", "characters": [{"id": "c"}], "prompt": "p"}, + model="wan3.0-video", + drop_params=False, + ) + + assert mapped == {} + + def test_parameters_block_is_merged(self): + mapped = DashScopeVideoConfig().map_openai_params( + video_create_optional_params={"parameters": {"prompt_extend": True, "audio": False}}, + model="wan3.0-video", + drop_params=False, + ) + + assert mapped == {"prompt_extend": True, "audio": False} + + def test_parameters_block_must_be_an_object(self): + with pytest.raises(ValueError, match="parameters must be an object"): + DashScopeVideoConfig().map_openai_params( + video_create_optional_params={"parameters": ["not", "an", "object"]}, + model="wan3.0-video", + drop_params=False, + ) + + @pytest.mark.parametrize("params", [{"seconds": "-1"}, {"parameters": {"duration": -1}}]) + def test_smart_duration_is_rejected_because_it_cannot_be_billed(self, params): + """duration -1 lets DashScope pick the length after the create call, + which is the only call litellm bills, so it would record $0 for a + video DashScope charges for.""" + mapped = DashScopeVideoConfig().map_openai_params( + video_create_optional_params=params, model="wan3.0-video", drop_params=False + ) + + with pytest.raises(UnsupportedParamsError, match="explicit duration") as raised: + _create(mapped) + + assert raised.value.status_code == 400 + + +class TestDashScopeVideoCreateResponse: + def test_task_id_is_encoded_with_provider_and_model(self): + """The create response only carries task_id, so litellm must wrap it for + later status/content calls to route back to dashscope.""" + video_obj = DashScopeVideoConfig().transform_video_create_response( + model="wan3.0-video", + raw_response=_mock_response( + {"output": {"task_status": "PENDING", "task_id": "0385dc79-5ff8"}, "request_id": "r1"} + ), + logging_obj=None, + custom_llm_provider="dashscope", + request_data={"model": "wan3.0-video", "parameters": {"duration": 10, "resolution": "720P"}}, + ) + + assert video_obj.status == "queued" + assert video_obj.model == "wan3.0-video" + decoded = decode_video_id_with_provider(video_obj.id) + assert decoded["custom_llm_provider"] == "dashscope" + assert decoded["model_id"] == "wan3.0-video" + assert decoded["video_id"] == "0385dc79-5ff8" + + def test_usage_carries_requested_cost_inputs_while_queued(self): + video_obj = DashScopeVideoConfig().transform_video_create_response( + model="wan3.0-video", + raw_response=_mock_response({"output": {"task_status": "PENDING", "task_id": "t1"}}), + logging_obj=None, + custom_llm_provider="dashscope", + request_data={"parameters": {"duration": 4, "resolution": "480P"}}, + ) + + assert video_obj.usage == {"duration_seconds": 4.0, "video_resolution": "480p"} + + @pytest.mark.parametrize( + "parameters,expected_usage", + [ + (None, {"duration_seconds": 5.0, "video_resolution": "1080p"}), + ({"resolution": "480P"}, {"duration_seconds": 5.0, "video_resolution": "480p"}), + ({"duration": 8}, {"duration_seconds": 8.0, "video_resolution": "1080p"}), + ], + ) + def test_omitted_duration_and_tier_bill_dashscope_defaults(self, parameters, expected_usage): + """DashScope fills an omitted duration (5s) and resolution (1080P) and + bills them; the create call is the billed one, so leaving either out + recorded the video at $0.""" + video_obj = DashScopeVideoConfig().transform_video_create_response( + model="wan3.0-video", + raw_response=_mock_response({"output": {"task_status": "PENDING", "task_id": "t1"}}), + logging_obj=None, + custom_llm_provider="dashscope", + request_data={"model": "wan3.0-video", **({"parameters": parameters} if parameters else {})}, + ) + + assert video_obj.usage == expected_usage + + def test_in_body_error_on_a_200_is_raised(self): + """DashScope reports create failures as a 200 with top-level code and no + output; without this the caller gets a queued video with an empty id.""" + with pytest.raises(DashScopeVideoError, match="No API-key provided"): + DashScopeVideoConfig().transform_video_create_response( + model="wan3.0-video", + raw_response=_mock_response( + {"code": "InvalidApiKey", "message": "No API-key provided.", "request_id": "r1"} + ), + logging_obj=None, + custom_llm_provider="dashscope", + request_data={}, + ) + + +class TestDashScopeVideoStatus: + SUCCEEDED = { + "request_id": "78c9b768", + "output": { + "task_id": "17ed7e50", + "task_status": "SUCCEEDED", + "submit_time": "2026-08-06 10:01:35.452", + "scheduled_time": "2026-08-06 10:01:35.507", + "end_time": "2026-08-06 10:13:33.838", + "orig_prompt": "a golden retriever", + "video_url": "https://oss.example.com/video.mp4", + }, + "usage": { + "video_count": 1, + "duration": 5.0, + "input_video_duration": 0.0, + "output_video_duration": 5.0, + "fps": 30, + "SR": 720, + "ratio": "16:9", + }, + } + + def test_succeeded_task_mapping(self): + video_obj = DashScopeVideoConfig().transform_video_status_retrieve_response( + raw_response=_mock_response(self.SUCCEEDED), + logging_obj=None, + custom_llm_provider="dashscope", + ) + + assert video_obj.status == "completed" + assert video_obj.seconds == "5.0" + assert video_obj.completed_at is not None + assert video_obj.created_at is not None + assert video_obj.completed_at > video_obj.created_at + assert decode_video_id_with_provider(video_obj.id)["video_id"] == "17ed7e50" + + def test_timestamps_are_read_as_utc_plus_8(self): + """DashScope stamps times in UTC+8 with no offset; reading them as UTC + would shift every video's created_at by 8 hours.""" + video_obj = DashScopeVideoConfig().transform_video_status_retrieve_response( + raw_response=_mock_response(self.SUCCEEDED), logging_obj=None, custom_llm_provider="dashscope" + ) + + submitted = datetime(2026, 8, 6, 10, 1, 35, 452000, tzinfo=timezone(timedelta(hours=8))) + assert video_obj.created_at == int(submitted.timestamp()) + + def test_delivered_resolution_beats_the_requested_one_for_billing(self): + """ratio adaptive lets DashScope deliver a different tier than asked + for; usage.SR is what was produced and therefore what is billed.""" + video_obj = DashScopeVideoConfig().transform_video_status_retrieve_response( + raw_response=_mock_response(self.SUCCEEDED), logging_obj=None, custom_llm_provider="dashscope" + ) + + assert video_obj.usage == {"duration_seconds": 5.0, "video_resolution": "720p"} + + def test_billed_duration_includes_input_video_seconds(self): + """For edit and extend, DashScope bills usage.duration (output plus + input video seconds), which is larger than the output alone.""" + payload = { + "output": {"task_id": "t1", "task_status": "SUCCEEDED", "video_url": "https://x/v.mp4"}, + "usage": {"duration": 12.0, "input_video_duration": 7.0, "output_video_duration": 5.0, "SR": 1080}, + } + video_obj = DashScopeVideoConfig().transform_video_status_retrieve_response( + raw_response=_mock_response(payload), logging_obj=None, custom_llm_provider="dashscope" + ) + + assert video_obj.usage["duration_seconds"] == 12.0 + assert video_obj.usage["video_resolution"] == "1080p" + + def test_size_is_rebuilt_from_shortest_side_and_ratio(self): + """DashScope splits geometry across SR and ratio; OpenAI's size field is + pixels, so reporting the bare ratio would be meaningless.""" + video_obj = DashScopeVideoConfig().transform_video_status_retrieve_response( + raw_response=_mock_response(self.SUCCEEDED), logging_obj=None, custom_llm_provider="dashscope" + ) + + assert video_obj.size == "1280x720" + + @pytest.mark.parametrize( + "task_status,expected", + [ + ("PENDING", "queued"), + ("RUNNING", "in_progress"), + ("SUCCEEDED", "completed"), + ("FAILED", "failed"), + ("CANCELED", "cancelled"), + ("UNKNOWN", "failed"), + ], + ) + def test_status_mapping(self, task_status, expected): + video_obj = DashScopeVideoConfig().transform_video_status_retrieve_response( + raw_response=_mock_response({"output": {"task_id": "t1", "task_status": task_status}}), + logging_obj=None, + custom_llm_provider="dashscope", + ) + + assert video_obj.status == expected + + def test_running_task_has_no_completion_time(self): + video_obj = DashScopeVideoConfig().transform_video_status_retrieve_response( + raw_response=_mock_response( + {"output": {"task_id": "t1", "task_status": "RUNNING", "submit_time": "2026-08-06 10:01:35.452"}} + ), + logging_obj=None, + custom_llm_provider="dashscope", + ) + + assert video_obj.completed_at is None + assert video_obj.usage is None + + def test_failed_task_maps_error_from_inside_output(self): + """Unlike creation, task failures nest code and message inside output.""" + video_obj = DashScopeVideoConfig().transform_video_status_retrieve_response( + raw_response=_mock_response( + { + "output": { + "task_id": "eff1443c", + "task_status": "FAILED", + "code": "InvalidParameter", + "message": "The two modes are mutually exclusive.", + } + } + ), + logging_obj=None, + custom_llm_provider="dashscope", + ) + + assert video_obj.status == "failed" + assert video_obj.error == { + "code": "InvalidParameter", + "message": "The two modes are mutually exclusive.", + } + + def test_polled_id_keeps_the_deployment_it_was_polled_with(self): + """The proxy resolves the deployment from the model encoded in the id; + dropping it from the returned id sent follow-up polls and downloads to + the default provider credentials instead of the deployment's.""" + polled_id = encode_video_id_with_provider("17ed7e50", "dashscope", "wan3-prod-deployment") + logging_obj = Mock(litellm_params={"video_id": polled_id}) + + video_obj = DashScopeVideoConfig().transform_video_status_retrieve_response( + raw_response=_mock_response(self.SUCCEEDED), logging_obj=logging_obj, custom_llm_provider="dashscope" + ) + + decoded = decode_video_id_with_provider(video_obj.id) + assert decoded["model_id"] == "wan3-prod-deployment" + assert decoded["video_id"] == "17ed7e50" + assert video_obj.model == "wan3-prod-deployment" + + def test_in_body_lookup_error_is_raised_instead_of_a_queued_video(self): + """An unknown or expired task id comes back as a 200 with top-level code + and no output; mapping it as a task produced a queued video with an + empty id that callers would poll forever.""" + with pytest.raises(DashScopeVideoError, match="task not found"): + DashScopeVideoConfig().transform_video_status_retrieve_response( + raw_response=_mock_response({"code": "InvalidParameter", "message": "task not found"}), + logging_obj=None, + custom_llm_provider="dashscope", + ) + + def test_request_decodes_task_id_from_wrapped_video_id(self): + encoded = encode_video_id_with_provider("17ed7e50", "dashscope", "wan3.0-video") + url, params = DashScopeVideoConfig().transform_video_status_retrieve_request( + video_id=encoded, api_base=API_BASE, litellm_params=GenericLiteLLMParams(), headers={} + ) + + assert url == f"{API_BASE}/api/v1/tasks/17ed7e50" + assert params == {} + + +class TestDashScopeVideoContent: + def test_content_request_targets_the_task_endpoint(self): + encoded = encode_video_id_with_provider("17ed7e50", "dashscope", "wan3.0-video") + url, params = DashScopeVideoConfig().transform_video_content_request( + video_id=encoded, api_base=API_BASE, litellm_params=GenericLiteLLMParams(), headers={} + ) + + assert url == f"{API_BASE}/api/v1/tasks/17ed7e50" + assert params == {} + + def test_url_is_extracted_from_a_succeeded_task(self): + task = _parse_task_response( + _mock_response( + {"output": {"task_id": "t1", "task_status": "SUCCEEDED", "video_url": "https://oss/video.mp4"}} + ) + ) + + assert _video_url_from_task(task) == "https://oss/video.mp4" + + @pytest.mark.parametrize("task_status", ["PENDING", "RUNNING"]) + def test_pending_task_raises_still_processing(self, task_status): + task = _parse_task_response(_mock_response({"output": {"task_id": "t1", "task_status": task_status}})) + + with pytest.raises(ValueError, match="still processing"): + _video_url_from_task(task) + + def test_failed_task_surfaces_the_upstream_message(self): + task = _parse_task_response( + _mock_response( + {"output": {"task_id": "t1", "task_status": "FAILED", "message": "content policy violation"}} + ) + ) + + with pytest.raises(ValueError, match="content policy violation"): + _video_url_from_task(task) + + def test_expired_task_explains_the_24h_window(self): + """A task id older than 24h comes back UNKNOWN with no error, which is + otherwise indistinguishable from a bad id.""" + task = _parse_task_response(_mock_response({"output": {"task_id": "t1", "task_status": "UNKNOWN"}})) + + with pytest.raises(ValueError, match="24 hours"): + _video_url_from_task(task) + + +class TestDashScopeVideoEnvironment: + def test_async_header_is_always_sent(self): + """DashScope rejects video-synthesis calls without X-DashScope-Async, + with 'current user api does not support synchronous calls'.""" + headers = DashScopeVideoConfig().validate_environment( + headers={}, model="wan3.0-video", api_key="sk-test", litellm_params=GenericLiteLLMParams() + ) + + assert headers["X-DashScope-Async"] == "enable" + assert headers["Authorization"] == "Bearer sk-test" + + def test_api_key_from_litellm_params(self): + headers = DashScopeVideoConfig().validate_environment( + headers={}, model="wan3.0-video", api_key=None, litellm_params=GenericLiteLLMParams(api_key="sk-params") + ) + + assert headers["Authorization"] == "Bearer sk-params" + + def test_missing_api_key_raises(self, monkeypatch): + monkeypatch.delenv("DASHSCOPE_API_KEY", raising=False) + monkeypatch.setattr("litellm.api_key", None) + + with pytest.raises(ValueError, match="DASHSCOPE_API_KEY"): + DashScopeVideoConfig().validate_environment( + headers={}, model="wan3.0-video", api_key=None, litellm_params=GenericLiteLLMParams() + ) + + def test_compatible_mode_chat_base_is_stripped_back_to_the_host(self): + """DASHSCOPE_API_BASE is shared with chat and points at + /compatible-mode/v1; the video API lives under /api/v1.""" + url = DashScopeVideoConfig().get_complete_url( + model="wan3.0-video", + api_base="https://dashscope.aliyuncs.com/compatible-mode/v1", + litellm_params={}, + ) + + assert url == API_BASE + + def test_workspace_scoped_host_is_preserved(self): + url = DashScopeVideoConfig().get_complete_url( + model="wan3.0-video", api_base="https://ws-123.cn-beijing.maas.aliyuncs.com/", litellm_params={} + ) + + assert url == "https://ws-123.cn-beijing.maas.aliyuncs.com" + + def test_default_host_is_beijing(self, monkeypatch): + monkeypatch.delenv("DASHSCOPE_API_BASE_VIDEO", raising=False) + + assert ( + DashScopeVideoConfig().get_complete_url(model="wan3.0-video", api_base=None, litellm_params={}) == API_BASE + ) + + +class TestDashScopeVideoBrandAliases: + def test_qwencloud_defaults_to_the_international_host(self, monkeypatch): + monkeypatch.delenv("QWENCLOUD_API_BASE_VIDEO", raising=False) + + assert ( + QwenCloudVideoConfig().get_complete_url(model="wan3.0-video", api_base=None, litellm_params={}) + == "https://dashscope-intl.aliyuncs.com" + ) + + def test_qwen_ai_platform_defaults_to_the_china_host(self, monkeypatch): + monkeypatch.delenv("QWEN_AI_PLATFORM_API_BASE_VIDEO", raising=False) + + assert ( + QwenAIPlatformVideoConfig().get_complete_url(model="wan3.0-video", api_base=None, litellm_params={}) + == API_BASE + ) + + def test_brand_alias_falls_back_to_the_shared_dashscope_key(self, monkeypatch): + monkeypatch.delenv("QWENCLOUD_API_KEY", raising=False) + monkeypatch.setenv("DASHSCOPE_API_KEY", "sk-shared") + monkeypatch.setattr("litellm.api_key", None) + + headers = QwenCloudVideoConfig().validate_environment( + headers={}, model="wan3.0-video", api_key=None, litellm_params=GenericLiteLLMParams() + ) + + assert headers["Authorization"] == "Bearer sk-shared" + + +class TestDashScopeVideoUnsupportedOperations: + @pytest.mark.parametrize("operation", ["remix", "list", "delete"]) + def test_operations_dashscope_does_not_expose_raise(self, operation): + """DashScope publishes no remix, list or delete surface, so these must + fail loudly instead of silently hitting a made-up endpoint.""" + config = DashScopeVideoConfig() + calls = { + "remix": lambda: config.transform_video_remix_request( + video_id="v", prompt="p", api_base=API_BASE, litellm_params=GenericLiteLLMParams(), headers={} + ), + "list": lambda: config.transform_video_list_request( + api_base=API_BASE, litellm_params=GenericLiteLLMParams(), headers={} + ), + "delete": lambda: config.transform_video_delete_request( + video_id="v", api_base=API_BASE, litellm_params=GenericLiteLLMParams(), headers={} + ), + } + + with pytest.raises(NotImplementedError, match="not supported by DashScope"): + calls[operation]() + + +class TestDashScopeVideoProviderWiring: + @pytest.mark.parametrize( + "provider,expected", + [ + ("dashscope", DashScopeVideoConfig), + ("qwencloud", QwenCloudVideoConfig), + ("qwen_ai_platform", QwenAIPlatformVideoConfig), + ], + ) + def test_provider_resolves_to_a_video_config(self, provider, expected): + """Without this registration litellm answers 'video generation is not + supported for dashscope' before any transform runs.""" + import litellm + from litellm.utils import ProviderConfigManager + + config = ProviderConfigManager.get_provider_video_config( + model="wan3.0-video", provider=litellm.LlmProviders(provider) + ) + + assert type(config) is expected + + +VIDEO_MODEL_TIERS: Final = ( + ("wan3.0-video", ("480p", "720p", "1080p")), + ("wan3.0-video-prime", ("480p", "720p", "1080p")), + ("happyhorse-1.1-t2v", ("480p", "720p", "1080p")), + ("happyhorse-1.1-i2v", ("480p", "720p", "1080p")), + ("happyhorse-1.1-r2v", ("480p", "720p", "1080p")), + ("happyhorse-1.0-t2v", ("720p", "1080p")), + ("happyhorse-1.0-i2v", ("720p", "1080p")), + ("happyhorse-1.0-r2v", ("720p", "1080p")), + ("wan2.7-t2v", ("720p", "1080p")), + ("wan2.7-i2v", ("720p", "1080p")), + ("wan2.7-r2v", ("720p", "1080p")), +) + + +@pytest.mark.usefixtures("local_model_cost_map") +class TestDashScopeVideoPricing: + @pytest.mark.parametrize("provider", ["dashscope", "qwencloud", "qwen_ai_platform"]) + @pytest.mark.parametrize("model,tiers", VIDEO_MODEL_TIERS) + def test_every_supported_tier_bills_its_own_rate(self, provider, model, tiers): + """A tier without its own rate falls back to the base rate, so a 480P + video would silently bill at the 1080P price; a missing alias entry + bills the whole video at zero.""" + from litellm import get_model_info + from litellm.llms.openai.cost_calculation import video_generation_cost + + info = get_model_info(model=model, custom_llm_provider=provider) + + for tier in tiers: + tier_rate = info[f"output_cost_per_second_{tier}"] + assert tier_rate > 0 + cost = video_generation_cost( + model=model, duration_seconds=5.0, custom_llm_provider=provider, video_resolution=tier + ) + assert cost == pytest.approx(tier_rate * 5.0) + + @pytest.mark.parametrize("provider", ["dashscope", "qwencloud", "qwen_ai_platform"]) + @pytest.mark.parametrize("model,tiers", VIDEO_MODEL_TIERS) + def test_higher_tiers_never_bill_less_and_1080p_is_the_default(self, provider, model, tiers): + """DashScope defaults to 1080P, so an unlabelled video must bill at the + 1080P rate rather than the cheapest tier.""" + from litellm import get_model_info + + info = get_model_info(model=model, custom_llm_provider=provider) + rates = [info[f"output_cost_per_second_{tier}"] for tier in tiers] + + assert rates == sorted(rates) + assert info["output_cost_per_second"] == info["output_cost_per_second_1080p"] + + @pytest.mark.parametrize("model,_", VIDEO_MODEL_TIERS) + def test_each_prefix_is_priced_for_the_region_its_default_host_serves(self, model, _): + """dashscope and qwen_ai_platform default to the Beijing host and + qwencloud to the international one, so the two Beijing-routed prefixes + must share a rate card.""" + from litellm import get_model_info + + dashscope = get_model_info(model=model, custom_llm_provider="dashscope") + qwen_ai_platform = get_model_info(model=model, custom_llm_provider="qwen_ai_platform") + + assert dashscope["output_cost_per_second_1080p"] == qwen_ai_platform["output_cost_per_second_1080p"] + + +class TestDashScopeVideoEndToEndRequest: + """ + Drive the real litellm entrypoint rather than calling the transform directly. + + ``video_generation()`` runs map_openai_params first and passes only the + *mapped* params to transform_video_create_request, so anything the mapper + drops never reaches the request builder. Asserting on the JSON the handler + actually serialized also catches body fragments that aren't JSON-encodable. + """ + + @staticmethod + async def _wire_request(**kwargs) -> dict: + sent: Final[list[httpx.Request]] = [] + + def respond(request: httpx.Request) -> httpx.Response: + sent.append(request) + return httpx.Response(200, json={"output": {"task_status": "PENDING", "task_id": "t1"}}) + + async with httpx.AsyncClient(transport=httpx.MockTransport(respond)) as http_client: + handler: Final = AsyncHTTPHandler() + await handler.close() + handler.client = http_client + await avideo_generation(api_key="sk-test", api_base=API_BASE, client=handler, **kwargs) + + (request,) = sent + return {"serialized": json.loads(request.content), "headers": dict(request.headers)} + + @pytest.mark.asyncio + async def test_input_reference_survives_param_mapping_into_the_body(self): + """Regression: map_openai_params used to drop input_reference, so an + image-to-video call silently degraded to text-to-video.""" + captured = await self._wire_request( + model="dashscope/wan3.0-video", + prompt="make it move", + input_reference="https://x/first.png", + seconds="10", + ) + + assert captured["serialized"]["input"]["media"] == [{"type": "first_frame", "url": "https://x/first.png"}] + assert captured["serialized"]["parameters"]["duration"] == 10 + assert "input_reference" not in captured["serialized"]["input"] + assert "input_reference" not in captured["serialized"].get("parameters", {}) + + @pytest.mark.asyncio + async def test_happyhorse_r2v_input_reference_reaches_the_body_as_a_reference_image(self): + captured = await self._wire_request( + model="dashscope/happyhorse-1.1-r2v", + prompt="make it move", + input_reference="https://x/subject.png", + ) + + assert captured["serialized"]["input"]["media"] == [{"type": "reference_image", "url": "https://x/subject.png"}] + + @pytest.mark.asyncio + async def test_file_input_reference_is_json_encodable(self): + """Regression: the first_frame entry was built as a MappingProxyType, + which json.dumps refuses, so every i2v request raised before sending.""" + captured = await self._wire_request( + model="dashscope/wan3.0-video", + prompt="make it move", + input_reference=io.BytesIO(PNG_BYTES), + ) + + media = captured["serialized"]["input"]["media"] + assert media[0]["type"] == "first_frame" + assert media[0]["url"].startswith("data:image/png;base64,") + + @pytest.mark.asyncio + async def test_auth_and_async_headers_reach_the_wire(self): + """DashScope rejects the call outright without X-DashScope-Async.""" + captured = await self._wire_request(model="dashscope/wan3.0-video", prompt="p") + + headers = {k.lower(): v for k, v in captured["headers"].items()} + assert headers["x-dashscope-async"] == "enable" + assert headers["authorization"] == "Bearer sk-test" + + @pytest.mark.asyncio + async def test_size_reaches_the_body_as_ratio_plus_resolution(self): + captured = await self._wire_request( + model="dashscope/happyhorse-1.1-t2v", prompt="a train", seconds="5", size="1920x1080" + ) + + assert captured["serialized"]["parameters"] == { + "resolution": "1080P", + "ratio": "16:9", + "duration": 5, + } + + @pytest.mark.asyncio + async def test_multi_role_media_array_reaches_the_body_intact(self): + media = [ + {"type": "reference_video", "url": "https://x/v.mp4"}, + {"type": "reference_audio", "url": "https://x/a.mp3"}, + ] + captured = await self._wire_request( + model="dashscope/wan3.0-video", prompt="视频1", media=media, duration=15, resolution="1080P" + ) + + assert captured["serialized"]["input"]["media"] == media + assert captured["serialized"]["parameters"]["duration"] == 15