feat(dashscope): add Wan 3.0, Wan 2.7 and HappyHorse video generation via /v1/videos

DashScopeVideoConfig maps litellm's OpenAI Videos API onto Alibaba Model
Studio's async task API: create hits
/api/v1/services/aigc/video-generation/video-synthesis and status and content
poll /api/v1/tasks/{task_id}. It is registered for dashscope and for the
qwencloud and qwen_ai_platform brand aliases, following the same
get_dashscope_family_*_config pattern image generation uses. Unlike the
Sora-shaped providers the request body is nested, so prompt and media go under
input and everything else goes under parameters. X-DashScope-Async is always
sent, because DashScope rejects the call outright without it

OpenAI params translate to DashScope's contract. seconds becomes duration, and
size yields both a ratio and a named resolution tier (1280x720 gives 16:9 plus
720P), since sending only the ratio leaves the request on the default 1080P and
silently bills at that rate. input_reference becomes a single media entry,
base64 encoded into a data URI for file inputs. Which media type a model takes
(first_frame or, on the reference-to-video models, reference_image) is read
from provider_specific_entry.dashscope_video_reference_input via
get_model_info rather than hardcoded; models without it predate the media
array and get the flat img_url field. A caller-supplied media array always
wins, as that is the only way to express multi-role combinations

The create call is the only billed call, so it bills what DashScope will
deliver: the requested duration and tier, or the documented 5 second and 1080P
defaults when they are omitted. Smart duration (duration -1) lets the service
pick the length after that call, so it is rejected rather than recorded at
zero, with a 400. Task creation and task lookup both report failures as a 200 carrying
top-level code and message with no output, so that shape is raised instead of
surfacing as a queued video with an empty id. Status responses re-encode the
deployment model from the polled id, so follow-up polls and downloads route to
the same deployment. Timestamps parse as UTC+8. DashScope publishes no list,
delete or remix endpoint, so those raise rather than guess at a URL

The end-to-end tests drive the real avideo_generation() entrypoint through an
httpx MockTransport and assert on the JSON body httpx actually encoded. That is
what catches a mapper that drops input_reference before the request is built,
or a body fragment json cannot serialize, neither of which a test calling the
transform directly would see

Both model_prices maps gain wan3.0-video, wan3.0-video-prime, wan2.7-t2v,
wan2.7-i2v, wan2.7-r2v and the HappyHorse 1.0 and 1.1 t2v, i2v and r2v models,
for dashscope and the qwencloud and qwen_ai_platform aliases, priced per output
second by resolution tier from Model Studio's USD rate cards. dashscope and
qwen_ai_platform default to the Beijing host and use its rate card; qwencloud
defaults to the international host and uses Singapore's. The three
*_API_BASE_VIDEO overrides are documented in BerriAI/litellm-docs#1920

Co-authored-by: AaronHowell <237895480@qq.com>
This commit is contained in:
Rick 2026-09-29 18:25:26 +08:00
parent 7f95b5f361
commit 80213ab516
9 changed files with 3221 additions and 0 deletions

View file

@ -16,6 +16,7 @@ if TYPE_CHECKING:
BaseImageGenerationConfig,
)
from litellm.llms.base_llm.rerank.transformation import BaseRerankConfig
from litellm.llms.base_llm.videos.transformation import BaseVideoConfig
DASHSCOPE_CHAT_COMPATIBLE_PATH: Final = "/compatible-mode/v1"
DASHSCOPE_RERANK_PATH: Final = "/compatible-api/v1/reranks"
@ -89,6 +90,22 @@ def get_dashscope_family_image_generation_config(
return DashScopeImageGenerationConfig()
def get_dashscope_family_video_config(
custom_llm_provider: str,
) -> "BaseVideoConfig":
if custom_llm_provider == "qwencloud":
from litellm.llms.dashscope.qwencloud import QwenCloudVideoConfig
return QwenCloudVideoConfig()
if custom_llm_provider == "qwen_ai_platform":
from litellm.llms.dashscope.qwen_ai_platform import QwenAIPlatformVideoConfig
return QwenAIPlatformVideoConfig()
from litellm.llms.dashscope.videos.transformation import DashScopeVideoConfig
return DashScopeVideoConfig()
def resolve_dashscope_family_api_key(custom_llm_provider: str, api_key: str | None) -> str | None:
if custom_llm_provider == "dashscope":
return api_key or get_secret_str("DASHSCOPE_API_KEY")

View file

@ -7,12 +7,14 @@ from .common_utils import resolve_dashscope_family_rerank_api_base
from .embed.transformation import DashScopeEmbeddingConfig
from .image_generation.transformation import DashScopeImageGenerationConfig
from .rerank.transformation import DashScopeRerankConfig
from .videos.transformation import DashScopeVideoConfig
QWEN_AI_PLATFORM_API_BASE: Final = "https://dashscope.aliyuncs.com/compatible-mode/v1"
QWEN_AI_PLATFORM_RERANK_API_BASE: Final = "https://dashscope.aliyuncs.com/compatible-api/v1/reranks"
QWEN_AI_PLATFORM_IMAGE_API_BASE: Final = (
"https://dashscope.aliyuncs.com/api/v1/services/aigc/multimodal-generation/generation"
)
QWEN_AI_PLATFORM_VIDEO_API_BASE: Final = "https://dashscope.aliyuncs.com"
def _resolve_qwen_ai_platform_api_key(api_key: str | None) -> str | None:
@ -63,3 +65,11 @@ class QwenAIPlatformImageGenerationConfig(DashScopeImageGenerationConfig):
def _resolve_image_api_base(self, image_api_base: str | None) -> str:
return image_api_base or get_secret_str("QWEN_AI_PLATFORM_API_BASE_IMAGE") or QWEN_AI_PLATFORM_IMAGE_API_BASE
class QwenAIPlatformVideoConfig(DashScopeVideoConfig):
def _resolve_api_key(self, api_key: str | None) -> str:
return _require_qwen_ai_platform_api_key(api_key)
def _resolve_video_api_base(self, video_api_base: str | None) -> str:
return video_api_base or get_secret_str("QWEN_AI_PLATFORM_API_BASE_VIDEO") or QWEN_AI_PLATFORM_VIDEO_API_BASE

View file

@ -7,12 +7,14 @@ from .common_utils import resolve_dashscope_family_rerank_api_base
from .embed.transformation import DashScopeEmbeddingConfig
from .image_generation.transformation import DashScopeImageGenerationConfig
from .rerank.transformation import DashScopeRerankConfig
from .videos.transformation import DashScopeVideoConfig
QWENCLOUD_API_BASE: Final = "https://dashscope-intl.aliyuncs.com/compatible-mode/v1"
QWENCLOUD_RERANK_API_BASE: Final = "https://dashscope-intl.aliyuncs.com/compatible-api/v1/reranks"
QWENCLOUD_IMAGE_API_BASE: Final = (
"https://dashscope-intl.aliyuncs.com/api/v1/services/aigc/multimodal-generation/generation"
)
QWENCLOUD_VIDEO_API_BASE: Final = "https://dashscope-intl.aliyuncs.com"
def _resolve_qwencloud_api_key(api_key: str | None) -> str | None:
@ -63,3 +65,11 @@ class QwenCloudImageGenerationConfig(DashScopeImageGenerationConfig):
def _resolve_image_api_base(self, image_api_base: str | None) -> str:
return image_api_base or get_secret_str("QWENCLOUD_API_BASE_IMAGE") or QWENCLOUD_IMAGE_API_BASE
class QwenCloudVideoConfig(DashScopeVideoConfig):
def _resolve_api_key(self, api_key: str | None) -> str:
return _require_qwencloud_api_key(api_key)
def _resolve_video_api_base(self, video_api_base: str | None) -> str:
return video_api_base or get_secret_str("QWENCLOUD_API_BASE_VIDEO") or QWENCLOUD_VIDEO_API_BASE

View file

@ -0,0 +1,706 @@
"""
DashScope (Alibaba Model Studio) async video task API: create, poll and download. DashScope has no list,
delete or remix endpoint, so those raise.
"""
import base64
from collections.abc import Mapping
from datetime import datetime
from io import BufferedReader, BytesIO
from math import gcd
from types import MappingProxyType
from typing import TYPE_CHECKING, Final
import httpx
from httpx._types import RequestFiles
from pydantic import BaseModel, TypeAdapter, ValidationError
import litellm
from litellm.exceptions import UnsupportedParamsError
from litellm.images.utils import ImageEditRequestUtils
from litellm.litellm_core_utils.url_utils import encode_url_path_segment
from litellm.llms.base_llm.chat.transformation import BaseLLMException
from litellm.llms.base_llm.videos.transformation import BaseVideoConfig
from litellm.llms.custom_httpx.http_handler import (
_get_httpx_client, # pyright: ignore[reportPrivateUsage, reportUnknownVariableType] # house cached-client factory has no public alias and its stub leaves params untyped
get_async_httpx_client, # pyright: ignore[reportUnknownVariableType] # factory stub leaves params untyped
)
from litellm.secret_managers.main import get_secret_str
from litellm.types.router import GenericLiteLLMParams
from litellm.types.videos.main import VideoCreateOptionalRequestParams, VideoObject
from litellm.types.videos.utils import (
decode_video_id_with_provider,
encode_video_id_with_provider,
extract_original_video_id,
)
from litellm.utils import get_model_info
if TYPE_CHECKING:
from litellm.litellm_core_utils.litellm_logging import Logging
from litellm.llms.custom_httpx.http_handler import HTTPHandler
class DashScopeVideoError(BaseLLMException):
pass
class _TaskOutput(BaseModel, frozen=True):
task_id: str = ""
task_status: str = ""
submit_time: str | None = None
scheduled_time: str | None = None
end_time: str | None = None
orig_prompt: str | None = None
video_url: str | None = None
code: str | None = None
message: str | None = None
class _TaskUsage(BaseModel, frozen=True):
video_count: int | None = None
duration: float | None = None
input_video_duration: float | None = None
output_video_duration: float | None = None
fps: int | None = None
SR: int | None = None
ratio: str | None = None
class _TaskResponse(BaseModel, frozen=True):
output: _TaskOutput | None = None
usage: _TaskUsage | None = None
request_id: str | None = None
code: str | None = None
message: str | None = None
DASHSCOPE_VIDEO_DEFAULT_API_BASE: Final = "https://dashscope.aliyuncs.com"
DASHSCOPE_VIDEO_SYNTHESIS_PATH: Final = "/api/v1/services/aigc/video-generation/video-synthesis"
DASHSCOPE_TASKS_PATH: Final = "/api/v1/tasks"
DASHSCOPE_COMPATIBLE_MODE_PATH: Final = "/compatible-mode/v1"
_EMPTY_PARAMS: Final[dict[str, object]] = {} # mutable-ok: BaseVideoConfig contract empty params
_TASK_RESPONSE_ADAPTER: Final = TypeAdapter(_TaskResponse)
_PARAMETERS_ADAPTER: Final = TypeAdapter(dict[str, object])
_STATUS_MAP: Final = MappingProxyType(
{
"PENDING": "queued",
"RUNNING": "in_progress",
"SUCCEEDED": "completed",
"FAILED": "failed",
"CANCELED": "cancelled",
"UNKNOWN": "failed",
}
)
_PENDING_STATUSES: Final = frozenset({"PENDING", "RUNNING"})
_LEGACY_INPUT_KEYS: Final = (
"img_url",
"first_frame_url",
"last_frame_url",
"audio_url",
)
_INPUT_KEYS: Final = ("media", "negative_prompt", *_LEGACY_INPUT_KEYS)
_REFERENCE_INPUT_FIELD: Final = "dashscope_video_reference_input"
_MEDIA_REFERENCE_TYPES: Final = frozenset({"first_frame", "reference_image"})
DASHSCOPE_DEFAULT_DURATION_SECONDS: Final = 5
DASHSCOPE_DEFAULT_RESOLUTION: Final = "1080P"
DASHSCOPE_SMART_DURATION: Final = -1
_PARAMETER_KEYS: Final = (
"resolution",
"ratio",
"duration",
"audio",
"seed",
"prompt_extend",
"watermark",
)
_DROP_FROM_REQUEST: Final = frozenset(
{
"model",
"prompt",
"user",
"characters",
"image",
"extra_headers",
"extra_query",
"extra_body",
}
)
_RESOLUTION_TIERS: Final = MappingProxyType({480: "480p", 720: "720p", 1080: "1080p"})
_RESOLUTION_HEIGHT_TIERS: Final[tuple[tuple[int, str], ...]] = (
(600, "480p"),
(900, "720p"),
)
def _parse_task_response(raw_response: httpx.Response) -> _TaskResponse:
return _TASK_RESPONSE_ADAPTER.validate_python(raw_response.json())
def _normalized_api_base(api_base: str) -> str:
trimmed: Final = api_base.rstrip("/")
if trimmed.endswith(DASHSCOPE_COMPATIBLE_MODE_PATH):
return trimmed[: -len(DASHSCOPE_COMPATIBLE_MODE_PATH)]
return trimmed
def _ratio_from_size(size: str) -> str | None:
if ":" in size:
return size
width, height = _size_dimensions(size) or (0, 0)
if not width or not height:
return None
divisor: Final = gcd(width, height)
return f"{width // divisor}:{height // divisor}"
def _size_dimensions(size: str) -> tuple[int, int] | None:
width_str, separator, height_str = size.partition("x")
if not separator or not (width_str.isdigit() and height_str.isdigit()):
return None
return int(width_str), int(height_str)
def _resolution_from_size(size: str) -> str | None:
"""
DashScope bills per second by named tier, so a size must map to its tier or it bills at the 1080P default.
"""
dimensions: Final = _size_dimensions(size)
if dimensions is None:
return None
shortest_side: Final = min(dimensions)
return next(
(label.upper() for threshold, label in _RESOLUTION_HEIGHT_TIERS if shortest_side < threshold),
"1080P",
)
def _duration_param(seconds: object) -> int | None:
if isinstance(seconds, bool):
return None
if isinstance(seconds, int):
return seconds
if not isinstance(seconds, str):
return None
try:
return int(float(seconds))
except ValueError:
return None
def _read_all_bytes(file_obj: object) -> bytes:
if isinstance(file_obj, (BytesIO, BufferedReader)):
current_position: Final = file_obj.tell()
file_obj.seek(0)
content: Final = file_obj.read()
file_obj.seek(current_position)
return content
if isinstance(file_obj, bytes):
return file_obj
if isinstance(file_obj, bytearray):
return bytes(file_obj)
read: Final = getattr(file_obj, "read", None)
if callable(read):
data: Final = read()
if isinstance(data, bytes):
return data
raise ValueError("input_reference must be a URL string, bytes, or a file object")
def _image_url(image: object) -> str:
if isinstance(image, str):
return image
content_type: Final = ImageEditRequestUtils.get_image_content_type(image)
encoded: Final = base64.b64encode(_read_all_bytes(image)).decode("utf-8")
return f"data:{content_type};base64,{encoded}"
def _media_reference_type(model: str) -> str | None:
"""Models without the field predate the media array and take the image on the flat ``img_url`` field."""
try:
info: Final = get_model_info(model=model, custom_llm_provider="dashscope")
except Exception:
return None
provider_specific: Final = info.get("provider_specific_entry")
value: Final = provider_specific.get(_REFERENCE_INPUT_FIELD) if isinstance(provider_specific, Mapping) else None
return value if isinstance(value, str) and value in _MEDIA_REFERENCE_TYPES else None
def _resolution_label(usage: _TaskUsage | None, requested_resolution: object) -> str | None:
if usage is not None and isinstance(usage.SR, int) and not isinstance(usage.SR, bool):
tier: Final = _RESOLUTION_TIERS.get(usage.SR)
if tier is not None:
return tier
if isinstance(requested_resolution, str) and requested_resolution.strip():
return requested_resolution.strip().lower()
return None
def _video_usage(usage: _TaskUsage | None, requested: Mapping[str, object]) -> Mapping[str, object]:
"""
``usage.duration`` is DashScope's billed duration, including input video seconds, so it beats the request.
"""
billed_duration: Final = usage.duration if usage is not None else None
requested_duration: Final = requested.get("duration")
duration_seconds: Final = (
float(billed_duration)
if isinstance(billed_duration, (int, float)) and not isinstance(billed_duration, bool)
else float(requested_duration)
if isinstance(requested_duration, (int, float))
and not isinstance(requested_duration, bool)
and requested_duration > 0
else None
)
resolution: Final = _resolution_label(usage, requested.get("resolution"))
return MappingProxyType(
{
key: value
for key, value in (("duration_seconds", duration_seconds), ("video_resolution", resolution))
if value is not None
}
)
def _timestamp(value: str | None) -> int | None:
"""
DashScope stamps times in UTC+8 with no offset.
"""
if not value:
return None
try:
parsed: Final = datetime.strptime(f"{value}+0800", "%Y-%m-%d %H:%M:%S.%f%z")
except ValueError:
return None
return int(parsed.timestamp())
def _error_block(output: _TaskOutput) -> Mapping[str, object] | None:
if not (output.code or output.message):
return None
return MappingProxyType(
{key: value for key, value in (("code", output.code), ("message", output.message)) if value is not None}
)
def _size_from_usage(usage: _TaskUsage | None) -> str | None:
if usage is None or not isinstance(usage.SR, int) or isinstance(usage.SR, bool) or usage.SR <= 0:
return None
if not usage.ratio or ":" not in usage.ratio:
return None
width_str, _, height_str = usage.ratio.partition(":")
if not (width_str.isdigit() and height_str.isdigit()):
return None
ratio_width: Final = int(width_str)
ratio_height: Final = int(height_str)
if not ratio_width or not ratio_height:
return None
shortest_ratio_side: Final = min(ratio_width, ratio_height)
return f"{usage.SR * ratio_width // shortest_ratio_side}x{usage.SR * ratio_height // shortest_ratio_side}"
def _video_object_from_task(
task: _TaskResponse,
model: str | None = None,
requested: Mapping[str, object] | None = None,
) -> VideoObject:
output: Final = task.output or _TaskOutput()
status: Final = _STATUS_MAP.get(output.task_status, "queued")
usage: Final = _video_usage(task.usage, requested or _EMPTY_PARAMS)
seconds: Final = usage.get("duration_seconds")
error_block: Final = _error_block(output)
return VideoObject(
id=output.task_id,
object="video",
status=status,
created_at=_timestamp(output.submit_time),
completed_at=_timestamp(output.end_time) if status == "completed" else None,
error=dict(error_block) if error_block is not None else None, # mutable-ok: VideoObject.error is a dict field
seconds=str(seconds) if seconds is not None else None,
size=_size_from_usage(task.usage),
model=model,
usage=dict(usage) if usage else None, # mutable-ok: VideoObject.usage is a dict field
)
def _video_url_from_task(task: _TaskResponse) -> str:
output: Final = task.output or _TaskOutput()
if output.video_url:
return output.video_url
if output.task_status in _PENDING_STATUSES:
raise ValueError(f"Video is still processing (status: {output.task_status}). Please wait and try again.")
if output.code or output.message:
raise ValueError(f"Video generation failed: {output.message or output.code}")
if output.task_status == "UNKNOWN":
raise ValueError("Task not found. DashScope task ids expire 24 hours after creation.")
raise ValueError("Video URL not found in task response. The task may not have succeeded yet.")
def _polled_model_id(logging_obj: object) -> str | None:
"""
The deployment the proxy routes by is the model encoded in the polled id, so it must survive into the returned id.
"""
litellm_params: Final = getattr(logging_obj, "litellm_params", None)
video_id: Final = litellm_params.get("video_id") if isinstance(litellm_params, Mapping) else None
if not isinstance(video_id, str):
return None
return decode_video_id_with_provider(video_id).get("model_id") or None
class DashScopeVideoConfig(BaseVideoConfig):
def get_supported_openai_params(self, model: str) -> list[str]: # mutable-ok: BaseVideoConfig contract returns list
return [ # mutable-ok: BaseVideoConfig contract returns list
"model",
"prompt",
"input_reference",
"seconds",
"size",
"user",
"extra_headers",
"media",
"resolution",
"ratio",
"duration",
"audio",
"seed",
"prompt_extend",
"watermark",
"negative_prompt",
"parameters",
*_LEGACY_INPUT_KEYS,
]
def map_openai_params(
self,
video_create_optional_params: VideoCreateOptionalRequestParams,
model: str,
drop_params: bool,
) -> dict[str, object]: # mutable-ok: BaseVideoConfig contract
mapped_params: Final[dict[str, object]] = {} # mutable-ok: BaseVideoConfig contract; extra_body merges into it
for key, value in video_create_optional_params.items():
if value is None or key in _DROP_FROM_REQUEST:
continue
if key == "seconds":
duration = _duration_param(value)
if duration is not None:
mapped_params["duration"] = duration
elif key == "size":
if not isinstance(value, str):
continue
ratio = _ratio_from_size(value)
if ratio is not None:
mapped_params.setdefault("ratio", ratio)
resolution = _resolution_from_size(value)
if resolution is not None:
mapped_params.setdefault("resolution", resolution)
elif key == "parameters":
try:
mapped_params.update(_PARAMETERS_ADAPTER.validate_python(value))
except ValidationError as e:
raise ValueError("parameters must be an object of DashScope request fields") from e
else:
mapped_params[key] = value
return mapped_params
def _resolve_api_key(self, api_key: str | None) -> str:
resolved_api_key: Final = api_key or get_secret_str("DASHSCOPE_API_KEY")
if resolved_api_key is None:
raise ValueError(
"DashScope API key is required. Set DASHSCOPE_API_KEY environment variable or pass api_key parameter."
)
return resolved_api_key
def _resolve_video_api_base(self, video_api_base: str | None) -> str:
return video_api_base or get_secret_str("DASHSCOPE_API_BASE_VIDEO") or DASHSCOPE_VIDEO_DEFAULT_API_BASE
def validate_environment(
self,
headers: dict[str, str], # mutable-ok: BaseVideoConfig contract; handler expects a mutable headers dict
model: str,
api_key: str | None = None,
litellm_params: GenericLiteLLMParams | None = None,
) -> dict[str, str]: # mutable-ok: BaseVideoConfig contract
resolved_api_key: Final = self._resolve_api_key(
api_key
or (litellm_params.api_key if litellm_params is not None and litellm_params.api_key else None)
or litellm.api_key
)
auth_headers: Final[dict[str, str]] = { # mutable-ok: httpx request headers are a mutable dict
"Authorization": f"Bearer {resolved_api_key}",
"Content-Type": "application/json",
"X-DashScope-Async": "enable",
}
headers.update(auth_headers)
return headers
def get_complete_url(
self,
model: str,
api_base: str | None,
litellm_params: dict[str, object], # mutable-ok: BaseVideoConfig contract
) -> str:
return _normalized_api_base(self._resolve_video_api_base(api_base))
def transform_video_create_request(
self,
model: str,
prompt: str,
api_base: str,
video_create_optional_request_params: dict[str, object], # mutable-ok: BaseVideoConfig contract
litellm_params: GenericLiteLLMParams,
headers: dict[str, str], # mutable-ok: BaseVideoConfig contract
) -> tuple[dict[str, object], RequestFiles, str]: # mutable-ok: BaseVideoConfig contract
reference_field, reference_value = self._reference_input(model, video_create_optional_request_params)
passthrough_input: Final = tuple(
(key, video_create_optional_request_params[key])
for key in _INPUT_KEYS
if video_create_optional_request_params.get(key) is not None
)
reference_entry: Final = (
((reference_field, reference_value),)
if reference_field is not None and reference_field not in frozenset(key for key, _ in passthrough_input)
else ()
)
request_input: Final[dict[str, object]] = dict( # mutable-ok: request body dict, JSON-serialized by the handler
(("prompt", prompt), *passthrough_input, *reference_entry)
)
parameters: Final[dict[str, object]] = dict( # mutable-ok: request body dict, JSON-serialized by the handler
(key, video_create_optional_request_params[key])
for key in _PARAMETER_KEYS
if video_create_optional_request_params.get(key) is not None
)
if parameters.get("duration") == DASHSCOPE_SMART_DURATION:
raise UnsupportedParamsError(
message=(
"Smart duration (duration=-1) is not supported through litellm: the video is billed when the task "
"is created, before DashScope has picked its length. Pass an explicit duration in seconds."
),
model=model,
llm_provider="dashscope",
)
request_data: Final[dict[str, object]] = { # mutable-ok: request body dict, JSON-serialized by the handler
"model": model,
"input": request_input,
}
if parameters:
request_data["parameters"] = parameters
return request_data, (), f"{api_base}{DASHSCOPE_VIDEO_SYNTHESIS_PATH}"
@staticmethod
def _reference_input(
model: str,
video_create_optional_request_params: Mapping[str, object],
) -> tuple[str | None, object]:
input_reference: Final = video_create_optional_request_params.get("input_reference")
if input_reference is None:
return None, None
image_url: Final = _image_url(input_reference)
media_type: Final = _media_reference_type(model)
if media_type is None:
return "img_url", image_url
# a plain dict because json.dumps cannot serialize a MappingProxyType
return "media", (dict[str, object](type=media_type, url=image_url),)
def transform_video_create_response(
self,
model: str,
raw_response: httpx.Response,
logging_obj: "Logging",
custom_llm_provider: str | None = None,
request_data: dict[str, object] | None = None, # mutable-ok: BaseVideoConfig contract
) -> VideoObject:
task: Final = _parse_task_response(raw_response)
self._raise_for_task_error(task, raw_response)
requested: Final = self._requested_parameters(request_data)
video_obj: Final = _video_object_from_task(task, model=model, requested=requested)
if custom_llm_provider and video_obj.id:
video_obj.id = encode_video_id_with_provider(video_obj.id, custom_llm_provider, model)
return video_obj
@staticmethod
def _requested_parameters(request_data: Mapping[str, object] | None) -> Mapping[str, object]:
"""
The create call is the only billed one, so an omitted duration or tier bills DashScope's defaults, not zero.
"""
parameters: Final = (request_data or _EMPTY_PARAMS).get("parameters")
requested: Final = (
_PARAMETERS_ADAPTER.validate_python(parameters) if isinstance(parameters, Mapping) else _EMPTY_PARAMS
)
return MappingProxyType(
{
"duration": DASHSCOPE_DEFAULT_DURATION_SECONDS,
"resolution": DASHSCOPE_DEFAULT_RESOLUTION,
**requested,
}
)
def _raise_for_task_error(self, task: _TaskResponse, raw_response: httpx.Response) -> None:
"""
Create and lookup both report failures as a 200 with top-level code and message and no output.
"""
if task.output is not None or not (task.code or task.message):
return
raise DashScopeVideoError(
status_code=raw_response.status_code,
message=task.message or task.code or "DashScope video request failed",
headers=raw_response.headers,
response=raw_response,
)
def transform_video_status_retrieve_request(
self,
video_id: str,
api_base: str,
litellm_params: GenericLiteLLMParams,
headers: dict[str, str], # mutable-ok: BaseVideoConfig contract
) -> tuple[str, dict[str, object]]: # mutable-ok: BaseVideoConfig contract
return self._task_url(video_id, api_base), _EMPTY_PARAMS
@staticmethod
def _task_url(video_id: str, api_base: str) -> str:
original_task_id: Final = extract_original_video_id(video_id)
encoded_task_id: Final = encode_url_path_segment(original_task_id, field_name="video_id")
return f"{api_base}{DASHSCOPE_TASKS_PATH}/{encoded_task_id}"
def transform_video_status_retrieve_response(
self,
raw_response: httpx.Response,
logging_obj: "Logging",
custom_llm_provider: str | None = None,
client: "HTTPHandler | None" = None,
) -> VideoObject:
task: Final = _parse_task_response(raw_response)
self._raise_for_task_error(task, raw_response)
model: Final = _polled_model_id(logging_obj)
video_obj: Final = _video_object_from_task(task, model=model)
if custom_llm_provider and video_obj.id:
video_obj.id = encode_video_id_with_provider(video_obj.id, custom_llm_provider, model)
return video_obj
def transform_video_content_request(
self,
video_id: str,
api_base: str,
litellm_params: GenericLiteLLMParams,
headers: dict[str, str], # mutable-ok: BaseVideoConfig contract
variant: str | None = None,
) -> tuple[str, dict[str, object]]: # mutable-ok: BaseVideoConfig contract
return self._task_url(video_id, api_base), _EMPTY_PARAMS
def transform_video_content_response(
self,
raw_response: httpx.Response,
logging_obj: "Logging",
) -> bytes:
video_url: Final = _video_url_from_task(_parse_task_response(raw_response))
httpx_client: Final = _get_httpx_client()
video_response: Final = httpx_client.get(video_url) # pyright: ignore[reportUnknownMemberType] # HTTPHandler.get stub leaves params/headers untyped
video_response.raise_for_status()
return video_response.content
async def async_transform_video_content_response(
self,
raw_response: httpx.Response,
logging_obj: "Logging",
) -> bytes:
video_url: Final = _video_url_from_task(_parse_task_response(raw_response))
async_httpx_client: Final = get_async_httpx_client(
llm_provider=litellm.LlmProviders.DASHSCOPE,
)
video_response: Final = await async_httpx_client.get(video_url) # pyright: ignore[reportUnknownMemberType] # HTTPHandler.get stub leaves params/headers untyped
video_response.raise_for_status()
return video_response.content
def transform_video_remix_request(
self,
video_id: str,
prompt: str,
api_base: str,
litellm_params: GenericLiteLLMParams,
headers: dict[str, str], # mutable-ok: BaseVideoConfig contract
extra_body: dict[str, object] | None = None, # mutable-ok: BaseVideoConfig contract
) -> tuple[str, dict[str, object]]: # mutable-ok: BaseVideoConfig contract
raise NotImplementedError(
"Video remix is not supported by DashScope. Wan 3.0 edits and extends a video by passing it back as a "
"reference_video media entry on video_generation() with an editing or extension prompt."
)
def transform_video_remix_response(
self,
raw_response: httpx.Response,
logging_obj: "Logging",
custom_llm_provider: str | None = None,
) -> VideoObject:
raise NotImplementedError("Video remix is not supported by DashScope.")
def transform_video_list_request(
self,
api_base: str,
litellm_params: GenericLiteLLMParams,
headers: dict[str, str], # mutable-ok: BaseVideoConfig contract
after: str | None = None,
limit: int | None = None,
order: str | None = None,
extra_query: dict[str, object] | None = None, # mutable-ok: BaseVideoConfig contract
) -> tuple[str, dict[str, object]]: # mutable-ok: BaseVideoConfig contract
raise NotImplementedError(
"Video list is not supported by DashScope. Retrieve tasks individually by task id within their 24 hour "
"retention window."
)
def transform_video_list_response(
self,
raw_response: httpx.Response,
logging_obj: "Logging",
custom_llm_provider: str | None = None,
) -> dict[str, str]: # mutable-ok: BaseVideoConfig contract
raise NotImplementedError("Video list is not supported by DashScope.")
def transform_video_delete_request(
self,
video_id: str,
api_base: str,
litellm_params: GenericLiteLLMParams,
headers: dict[str, str], # mutable-ok: BaseVideoConfig contract
) -> tuple[str, dict[str, object]]: # mutable-ok: BaseVideoConfig contract
raise NotImplementedError(
"Video delete is not supported by DashScope. Tasks and their artifacts expire 24 hours after creation."
)
def transform_video_delete_response(
self,
raw_response: httpx.Response,
logging_obj: "Logging",
) -> VideoObject:
raise NotImplementedError("Video delete is not supported by DashScope.")
def get_error_class(
self,
error_message: str,
status_code: int,
headers: dict[str, str] | httpx.Headers, # mutable-ok: BaseVideoConfig contract
) -> BaseLLMException:
return DashScopeVideoError(
status_code=status_code,
message=error_message,
headers=headers,
)

View file

@ -17019,6 +17019,281 @@
"/v1/images/generations"
]
},
"dashscope/wan3.0-video": {
"litellm_provider": "dashscope",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/wan3-0-video",
"output_cost_per_second": 0.165025,
"output_cost_per_second_480p": 0.041256,
"output_cost_per_second_720p": 0.082513,
"output_cost_per_second_1080p": 0.165025,
"supported_modalities": [
"text",
"image",
"video",
"audio"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "Wan 3.0 all-in-one video generation (text, first/last frame, multimodal reference, edit, extend). Billed duration is usage.duration, which adds input video seconds for the edit and extend modes. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"dashscope/wan3.0-video-prime": {
"litellm_provider": "dashscope",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/wan3-0-video-prime",
"output_cost_per_second": 0.254399,
"output_cost_per_second_480p": 0.0636,
"output_cost_per_second_720p": 0.127199,
"output_cost_per_second_1080p": 0.254399,
"supported_modalities": [
"text",
"image",
"video",
"audio"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "Wan 3.0 fast tier. Billed duration is usage.duration, which adds input video seconds for the edit and extend modes. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"dashscope/happyhorse-1.1-t2v": {
"litellm_provider": "dashscope",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-t2v",
"output_cost_per_second": 0.165026,
"output_cost_per_second_480p": 0.0618845,
"output_cost_per_second_720p": 0.123769,
"output_cost_per_second_1080p": 0.165026,
"supported_modalities": [
"text"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "HappyHorse 1.1 text-to-video. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"dashscope/happyhorse-1.1-i2v": {
"litellm_provider": "dashscope",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-i2v",
"output_cost_per_second": 0.165026,
"output_cost_per_second_480p": 0.0618845,
"output_cost_per_second_720p": 0.123769,
"output_cost_per_second_1080p": 0.165026,
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "HappyHorse 1.1 image-to-video from a first frame. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"dashscope/happyhorse-1.1-r2v": {
"litellm_provider": "dashscope",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-r2v",
"output_cost_per_second": 0.165026,
"output_cost_per_second_480p": 0.0618845,
"output_cost_per_second_720p": 0.123769,
"output_cost_per_second_1080p": 0.165026,
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "reference_image"
},
"metadata": {
"comment": "HappyHorse 1.1 reference-to-video, 1 to 9 reference images. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"dashscope/happyhorse-1.0-t2v": {
"litellm_provider": "dashscope",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-t2v",
"output_cost_per_second": 0.220034,
"output_cost_per_second_720p": 0.123769,
"output_cost_per_second_1080p": 0.220034,
"supported_modalities": [
"text"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "HappyHorse 1.0 text-to-video. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"dashscope/happyhorse-1.0-i2v": {
"litellm_provider": "dashscope",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-i2v",
"output_cost_per_second": 0.220034,
"output_cost_per_second_720p": 0.123769,
"output_cost_per_second_1080p": 0.220034,
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "HappyHorse 1.0 image-to-video from a first frame. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"dashscope/happyhorse-1.0-r2v": {
"litellm_provider": "dashscope",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-r2v",
"output_cost_per_second": 0.220034,
"output_cost_per_second_720p": 0.123769,
"output_cost_per_second_1080p": 0.220034,
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "reference_image"
},
"metadata": {
"comment": "HappyHorse 1.0 reference-to-video. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"dashscope/wan2.7-t2v": {
"litellm_provider": "dashscope",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-t2v",
"output_cost_per_second": 0.143353,
"output_cost_per_second_720p": 0.086012,
"output_cost_per_second_1080p": 0.143353,
"supported_modalities": [
"text",
"audio"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "Wan 2.7 text-to-video. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"dashscope/wan2.7-i2v": {
"litellm_provider": "dashscope",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-i2v",
"output_cost_per_second": 0.143353,
"output_cost_per_second_720p": 0.086012,
"output_cost_per_second_1080p": 0.143353,
"supported_modalities": [
"text",
"image",
"video",
"audio"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "Wan 2.7 image-to-video (first frame, first and last frame, continuation). No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"dashscope/wan2.7-r2v": {
"litellm_provider": "dashscope",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-r2v",
"output_cost_per_second": 0.143353,
"output_cost_per_second_720p": 0.086012,
"output_cost_per_second_1080p": 0.143353,
"supported_modalities": [
"text",
"image",
"video",
"audio"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "reference_image"
},
"metadata": {
"comment": "Wan 2.7 reference-to-video. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwencloud/deepseek-v4-flash": {
"cache_read_input_token_cost": 4e-08,
"input_cost_per_token": 2e-07,
@ -17971,6 +18246,281 @@
"/v1/images/generations"
]
},
"qwencloud/wan3.0-video": {
"litellm_provider": "qwencloud",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/wan3-0-video",
"output_cost_per_second": 0.2,
"output_cost_per_second_480p": 0.05,
"output_cost_per_second_720p": 0.1,
"output_cost_per_second_1080p": 0.2,
"supported_modalities": [
"text",
"image",
"video",
"audio"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "Wan 3.0 all-in-one video generation (text, first/last frame, multimodal reference, edit, extend). Billed duration is usage.duration, which adds input video seconds for the edit and extend modes. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwencloud/wan3.0-video-prime": {
"litellm_provider": "qwencloud",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/wan3-0-video-prime",
"output_cost_per_second": 0.28,
"output_cost_per_second_480p": 0.068,
"output_cost_per_second_720p": 0.14,
"output_cost_per_second_1080p": 0.28,
"supported_modalities": [
"text",
"image",
"video",
"audio"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "Wan 3.0 fast tier. Billed duration is usage.duration, which adds input video seconds for the edit and extend modes. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwencloud/happyhorse-1.1-t2v": {
"litellm_provider": "qwencloud",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-t2v",
"output_cost_per_second": 0.18,
"output_cost_per_second_480p": 0.07,
"output_cost_per_second_720p": 0.14,
"output_cost_per_second_1080p": 0.18,
"supported_modalities": [
"text"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "HappyHorse 1.1 text-to-video. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwencloud/happyhorse-1.1-i2v": {
"litellm_provider": "qwencloud",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-i2v",
"output_cost_per_second": 0.18,
"output_cost_per_second_480p": 0.07,
"output_cost_per_second_720p": 0.14,
"output_cost_per_second_1080p": 0.18,
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "HappyHorse 1.1 image-to-video from a first frame. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwencloud/happyhorse-1.1-r2v": {
"litellm_provider": "qwencloud",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-r2v",
"output_cost_per_second": 0.18,
"output_cost_per_second_480p": 0.07,
"output_cost_per_second_720p": 0.14,
"output_cost_per_second_1080p": 0.18,
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "reference_image"
},
"metadata": {
"comment": "HappyHorse 1.1 reference-to-video, 1 to 9 reference images. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwencloud/happyhorse-1.0-t2v": {
"litellm_provider": "qwencloud",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-t2v",
"output_cost_per_second": 0.24,
"output_cost_per_second_720p": 0.14,
"output_cost_per_second_1080p": 0.24,
"supported_modalities": [
"text"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "HappyHorse 1.0 text-to-video. No 480P tier. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwencloud/happyhorse-1.0-i2v": {
"litellm_provider": "qwencloud",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-i2v",
"output_cost_per_second": 0.24,
"output_cost_per_second_720p": 0.14,
"output_cost_per_second_1080p": 0.24,
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "HappyHorse 1.0 image-to-video from a first frame. No 480P tier. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwencloud/happyhorse-1.0-r2v": {
"litellm_provider": "qwencloud",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-r2v",
"output_cost_per_second": 0.24,
"output_cost_per_second_720p": 0.14,
"output_cost_per_second_1080p": 0.24,
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "reference_image"
},
"metadata": {
"comment": "HappyHorse 1.0 reference-to-video. No 480P tier. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwencloud/wan2.7-t2v": {
"litellm_provider": "qwencloud",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-t2v",
"output_cost_per_second": 0.15,
"output_cost_per_second_720p": 0.1,
"output_cost_per_second_1080p": 0.15,
"supported_modalities": [
"text",
"audio"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "Wan 2.7 text-to-video. No 480P tier. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwencloud/wan2.7-i2v": {
"litellm_provider": "qwencloud",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-i2v",
"output_cost_per_second": 0.15,
"output_cost_per_second_720p": 0.1,
"output_cost_per_second_1080p": 0.15,
"supported_modalities": [
"text",
"image",
"video",
"audio"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "Wan 2.7 image-to-video (first frame, first and last frame, continuation). No 480P tier. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwencloud/wan2.7-r2v": {
"litellm_provider": "qwencloud",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-r2v",
"output_cost_per_second": 0.15,
"output_cost_per_second_720p": 0.1,
"output_cost_per_second_1080p": 0.15,
"supported_modalities": [
"text",
"image",
"video",
"audio"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "reference_image"
},
"metadata": {
"comment": "Wan 2.7 reference-to-video. No 480P tier. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwen_ai_platform/deepseek-v4-flash": {
"cache_read_input_token_cost": 4e-08,
"input_cost_per_token": 2e-07,
@ -18963,6 +19513,281 @@
"/v1/images/generations"
]
},
"qwen_ai_platform/wan3.0-video": {
"litellm_provider": "qwen_ai_platform",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/wan3-0-video",
"output_cost_per_second": 0.165025,
"output_cost_per_second_480p": 0.041256,
"output_cost_per_second_720p": 0.082513,
"output_cost_per_second_1080p": 0.165025,
"supported_modalities": [
"text",
"image",
"video",
"audio"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "Wan 3.0 all-in-one video generation (text, first/last frame, multimodal reference, edit, extend). Billed duration is usage.duration, which adds input video seconds for the edit and extend modes. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwen_ai_platform/wan3.0-video-prime": {
"litellm_provider": "qwen_ai_platform",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/wan3-0-video-prime",
"output_cost_per_second": 0.254399,
"output_cost_per_second_480p": 0.0636,
"output_cost_per_second_720p": 0.127199,
"output_cost_per_second_1080p": 0.254399,
"supported_modalities": [
"text",
"image",
"video",
"audio"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "Wan 3.0 fast tier. Billed duration is usage.duration, which adds input video seconds for the edit and extend modes. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwen_ai_platform/happyhorse-1.1-t2v": {
"litellm_provider": "qwen_ai_platform",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-t2v",
"output_cost_per_second": 0.165026,
"output_cost_per_second_480p": 0.0618845,
"output_cost_per_second_720p": 0.123769,
"output_cost_per_second_1080p": 0.165026,
"supported_modalities": [
"text"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "HappyHorse 1.1 text-to-video. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwen_ai_platform/happyhorse-1.1-i2v": {
"litellm_provider": "qwen_ai_platform",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-i2v",
"output_cost_per_second": 0.165026,
"output_cost_per_second_480p": 0.0618845,
"output_cost_per_second_720p": 0.123769,
"output_cost_per_second_1080p": 0.165026,
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "HappyHorse 1.1 image-to-video from a first frame. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwen_ai_platform/happyhorse-1.1-r2v": {
"litellm_provider": "qwen_ai_platform",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-r2v",
"output_cost_per_second": 0.165026,
"output_cost_per_second_480p": 0.0618845,
"output_cost_per_second_720p": 0.123769,
"output_cost_per_second_1080p": 0.165026,
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "reference_image"
},
"metadata": {
"comment": "HappyHorse 1.1 reference-to-video, 1 to 9 reference images. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwen_ai_platform/happyhorse-1.0-t2v": {
"litellm_provider": "qwen_ai_platform",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-t2v",
"output_cost_per_second": 0.220034,
"output_cost_per_second_720p": 0.123769,
"output_cost_per_second_1080p": 0.220034,
"supported_modalities": [
"text"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "HappyHorse 1.0 text-to-video. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwen_ai_platform/happyhorse-1.0-i2v": {
"litellm_provider": "qwen_ai_platform",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-i2v",
"output_cost_per_second": 0.220034,
"output_cost_per_second_720p": 0.123769,
"output_cost_per_second_1080p": 0.220034,
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "HappyHorse 1.0 image-to-video from a first frame. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwen_ai_platform/happyhorse-1.0-r2v": {
"litellm_provider": "qwen_ai_platform",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-r2v",
"output_cost_per_second": 0.220034,
"output_cost_per_second_720p": 0.123769,
"output_cost_per_second_1080p": 0.220034,
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "reference_image"
},
"metadata": {
"comment": "HappyHorse 1.0 reference-to-video. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwen_ai_platform/wan2.7-t2v": {
"litellm_provider": "qwen_ai_platform",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-t2v",
"output_cost_per_second": 0.143353,
"output_cost_per_second_720p": 0.086012,
"output_cost_per_second_1080p": 0.143353,
"supported_modalities": [
"text",
"audio"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "Wan 2.7 text-to-video. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwen_ai_platform/wan2.7-i2v": {
"litellm_provider": "qwen_ai_platform",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-i2v",
"output_cost_per_second": 0.143353,
"output_cost_per_second_720p": 0.086012,
"output_cost_per_second_1080p": 0.143353,
"supported_modalities": [
"text",
"image",
"video",
"audio"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "Wan 2.7 image-to-video (first frame, first and last frame, continuation). No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwen_ai_platform/wan2.7-r2v": {
"litellm_provider": "qwen_ai_platform",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-r2v",
"output_cost_per_second": 0.143353,
"output_cost_per_second_720p": 0.086012,
"output_cost_per_second_1080p": 0.143353,
"supported_modalities": [
"text",
"image",
"video",
"audio"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "reference_image"
},
"metadata": {
"comment": "Wan 2.7 reference-to-video. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"databricks/databricks-bge-large-en": {
"cache_creation_input_token_cost": 1.0003e-07,
"cache_read_input_token_cost": 1.0003e-07,

View file

@ -9626,6 +9626,16 @@ class ProviderConfigManager:
from litellm.llms.vertex_ai.videos.transformation import VertexAIVideoConfig
return VertexAIVideoConfig()
elif provider in (
LlmProviders.DASHSCOPE,
LlmProviders.QWENCLOUD,
LlmProviders.QWEN_AI_PLATFORM,
):
from litellm.llms.dashscope.common_utils import (
get_dashscope_family_video_config,
)
return get_dashscope_family_video_config(provider.value)
elif LlmProviders.RUNWAYML == provider:
from litellm.llms.runwayml.videos.transformation import RunwayMLVideoConfig

View file

@ -17019,6 +17019,281 @@
"/v1/images/generations"
]
},
"dashscope/wan3.0-video": {
"litellm_provider": "dashscope",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/wan3-0-video",
"output_cost_per_second": 0.165025,
"output_cost_per_second_480p": 0.041256,
"output_cost_per_second_720p": 0.082513,
"output_cost_per_second_1080p": 0.165025,
"supported_modalities": [
"text",
"image",
"video",
"audio"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "Wan 3.0 all-in-one video generation (text, first/last frame, multimodal reference, edit, extend). Billed duration is usage.duration, which adds input video seconds for the edit and extend modes. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"dashscope/wan3.0-video-prime": {
"litellm_provider": "dashscope",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/wan3-0-video-prime",
"output_cost_per_second": 0.254399,
"output_cost_per_second_480p": 0.0636,
"output_cost_per_second_720p": 0.127199,
"output_cost_per_second_1080p": 0.254399,
"supported_modalities": [
"text",
"image",
"video",
"audio"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "Wan 3.0 fast tier. Billed duration is usage.duration, which adds input video seconds for the edit and extend modes. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"dashscope/happyhorse-1.1-t2v": {
"litellm_provider": "dashscope",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-t2v",
"output_cost_per_second": 0.165026,
"output_cost_per_second_480p": 0.0618845,
"output_cost_per_second_720p": 0.123769,
"output_cost_per_second_1080p": 0.165026,
"supported_modalities": [
"text"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "HappyHorse 1.1 text-to-video. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"dashscope/happyhorse-1.1-i2v": {
"litellm_provider": "dashscope",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-i2v",
"output_cost_per_second": 0.165026,
"output_cost_per_second_480p": 0.0618845,
"output_cost_per_second_720p": 0.123769,
"output_cost_per_second_1080p": 0.165026,
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "HappyHorse 1.1 image-to-video from a first frame. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"dashscope/happyhorse-1.1-r2v": {
"litellm_provider": "dashscope",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-r2v",
"output_cost_per_second": 0.165026,
"output_cost_per_second_480p": 0.0618845,
"output_cost_per_second_720p": 0.123769,
"output_cost_per_second_1080p": 0.165026,
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "reference_image"
},
"metadata": {
"comment": "HappyHorse 1.1 reference-to-video, 1 to 9 reference images. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"dashscope/happyhorse-1.0-t2v": {
"litellm_provider": "dashscope",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-t2v",
"output_cost_per_second": 0.220034,
"output_cost_per_second_720p": 0.123769,
"output_cost_per_second_1080p": 0.220034,
"supported_modalities": [
"text"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "HappyHorse 1.0 text-to-video. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"dashscope/happyhorse-1.0-i2v": {
"litellm_provider": "dashscope",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-i2v",
"output_cost_per_second": 0.220034,
"output_cost_per_second_720p": 0.123769,
"output_cost_per_second_1080p": 0.220034,
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "HappyHorse 1.0 image-to-video from a first frame. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"dashscope/happyhorse-1.0-r2v": {
"litellm_provider": "dashscope",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-r2v",
"output_cost_per_second": 0.220034,
"output_cost_per_second_720p": 0.123769,
"output_cost_per_second_1080p": 0.220034,
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "reference_image"
},
"metadata": {
"comment": "HappyHorse 1.0 reference-to-video. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"dashscope/wan2.7-t2v": {
"litellm_provider": "dashscope",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-t2v",
"output_cost_per_second": 0.143353,
"output_cost_per_second_720p": 0.086012,
"output_cost_per_second_1080p": 0.143353,
"supported_modalities": [
"text",
"audio"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "Wan 2.7 text-to-video. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"dashscope/wan2.7-i2v": {
"litellm_provider": "dashscope",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-i2v",
"output_cost_per_second": 0.143353,
"output_cost_per_second_720p": 0.086012,
"output_cost_per_second_1080p": 0.143353,
"supported_modalities": [
"text",
"image",
"video",
"audio"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "Wan 2.7 image-to-video (first frame, first and last frame, continuation). No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"dashscope/wan2.7-r2v": {
"litellm_provider": "dashscope",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-r2v",
"output_cost_per_second": 0.143353,
"output_cost_per_second_720p": 0.086012,
"output_cost_per_second_1080p": 0.143353,
"supported_modalities": [
"text",
"image",
"video",
"audio"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "reference_image"
},
"metadata": {
"comment": "Wan 2.7 reference-to-video. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwencloud/deepseek-v4-flash": {
"cache_read_input_token_cost": 4e-08,
"input_cost_per_token": 2e-07,
@ -17971,6 +18246,281 @@
"/v1/images/generations"
]
},
"qwencloud/wan3.0-video": {
"litellm_provider": "qwencloud",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/wan3-0-video",
"output_cost_per_second": 0.2,
"output_cost_per_second_480p": 0.05,
"output_cost_per_second_720p": 0.1,
"output_cost_per_second_1080p": 0.2,
"supported_modalities": [
"text",
"image",
"video",
"audio"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "Wan 3.0 all-in-one video generation (text, first/last frame, multimodal reference, edit, extend). Billed duration is usage.duration, which adds input video seconds for the edit and extend modes. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwencloud/wan3.0-video-prime": {
"litellm_provider": "qwencloud",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/wan3-0-video-prime",
"output_cost_per_second": 0.28,
"output_cost_per_second_480p": 0.068,
"output_cost_per_second_720p": 0.14,
"output_cost_per_second_1080p": 0.28,
"supported_modalities": [
"text",
"image",
"video",
"audio"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "Wan 3.0 fast tier. Billed duration is usage.duration, which adds input video seconds for the edit and extend modes. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwencloud/happyhorse-1.1-t2v": {
"litellm_provider": "qwencloud",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-t2v",
"output_cost_per_second": 0.18,
"output_cost_per_second_480p": 0.07,
"output_cost_per_second_720p": 0.14,
"output_cost_per_second_1080p": 0.18,
"supported_modalities": [
"text"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "HappyHorse 1.1 text-to-video. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwencloud/happyhorse-1.1-i2v": {
"litellm_provider": "qwencloud",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-i2v",
"output_cost_per_second": 0.18,
"output_cost_per_second_480p": 0.07,
"output_cost_per_second_720p": 0.14,
"output_cost_per_second_1080p": 0.18,
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "HappyHorse 1.1 image-to-video from a first frame. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwencloud/happyhorse-1.1-r2v": {
"litellm_provider": "qwencloud",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-r2v",
"output_cost_per_second": 0.18,
"output_cost_per_second_480p": 0.07,
"output_cost_per_second_720p": 0.14,
"output_cost_per_second_1080p": 0.18,
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "reference_image"
},
"metadata": {
"comment": "HappyHorse 1.1 reference-to-video, 1 to 9 reference images. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwencloud/happyhorse-1.0-t2v": {
"litellm_provider": "qwencloud",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-t2v",
"output_cost_per_second": 0.24,
"output_cost_per_second_720p": 0.14,
"output_cost_per_second_1080p": 0.24,
"supported_modalities": [
"text"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "HappyHorse 1.0 text-to-video. No 480P tier. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwencloud/happyhorse-1.0-i2v": {
"litellm_provider": "qwencloud",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-i2v",
"output_cost_per_second": 0.24,
"output_cost_per_second_720p": 0.14,
"output_cost_per_second_1080p": 0.24,
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "HappyHorse 1.0 image-to-video from a first frame. No 480P tier. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwencloud/happyhorse-1.0-r2v": {
"litellm_provider": "qwencloud",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-r2v",
"output_cost_per_second": 0.24,
"output_cost_per_second_720p": 0.14,
"output_cost_per_second_1080p": 0.24,
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "reference_image"
},
"metadata": {
"comment": "HappyHorse 1.0 reference-to-video. No 480P tier. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwencloud/wan2.7-t2v": {
"litellm_provider": "qwencloud",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-t2v",
"output_cost_per_second": 0.15,
"output_cost_per_second_720p": 0.1,
"output_cost_per_second_1080p": 0.15,
"supported_modalities": [
"text",
"audio"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "Wan 2.7 text-to-video. No 480P tier. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwencloud/wan2.7-i2v": {
"litellm_provider": "qwencloud",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-i2v",
"output_cost_per_second": 0.15,
"output_cost_per_second_720p": 0.1,
"output_cost_per_second_1080p": 0.15,
"supported_modalities": [
"text",
"image",
"video",
"audio"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "Wan 2.7 image-to-video (first frame, first and last frame, continuation). No 480P tier. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwencloud/wan2.7-r2v": {
"litellm_provider": "qwencloud",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-r2v",
"output_cost_per_second": 0.15,
"output_cost_per_second_720p": 0.1,
"output_cost_per_second_1080p": 0.15,
"supported_modalities": [
"text",
"image",
"video",
"audio"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "reference_image"
},
"metadata": {
"comment": "Wan 2.7 reference-to-video. No 480P tier. Singapore list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwen_ai_platform/deepseek-v4-flash": {
"cache_read_input_token_cost": 4e-08,
"input_cost_per_token": 2e-07,
@ -18963,6 +19513,281 @@
"/v1/images/generations"
]
},
"qwen_ai_platform/wan3.0-video": {
"litellm_provider": "qwen_ai_platform",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/wan3-0-video",
"output_cost_per_second": 0.165025,
"output_cost_per_second_480p": 0.041256,
"output_cost_per_second_720p": 0.082513,
"output_cost_per_second_1080p": 0.165025,
"supported_modalities": [
"text",
"image",
"video",
"audio"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "Wan 3.0 all-in-one video generation (text, first/last frame, multimodal reference, edit, extend). Billed duration is usage.duration, which adds input video seconds for the edit and extend modes. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwen_ai_platform/wan3.0-video-prime": {
"litellm_provider": "qwen_ai_platform",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/wan3-0-video-prime",
"output_cost_per_second": 0.254399,
"output_cost_per_second_480p": 0.0636,
"output_cost_per_second_720p": 0.127199,
"output_cost_per_second_1080p": 0.254399,
"supported_modalities": [
"text",
"image",
"video",
"audio"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "Wan 3.0 fast tier. Billed duration is usage.duration, which adds input video seconds for the edit and extend modes. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwen_ai_platform/happyhorse-1.1-t2v": {
"litellm_provider": "qwen_ai_platform",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-t2v",
"output_cost_per_second": 0.165026,
"output_cost_per_second_480p": 0.0618845,
"output_cost_per_second_720p": 0.123769,
"output_cost_per_second_1080p": 0.165026,
"supported_modalities": [
"text"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "HappyHorse 1.1 text-to-video. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwen_ai_platform/happyhorse-1.1-i2v": {
"litellm_provider": "qwen_ai_platform",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-i2v",
"output_cost_per_second": 0.165026,
"output_cost_per_second_480p": 0.0618845,
"output_cost_per_second_720p": 0.123769,
"output_cost_per_second_1080p": 0.165026,
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "HappyHorse 1.1 image-to-video from a first frame. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwen_ai_platform/happyhorse-1.1-r2v": {
"litellm_provider": "qwen_ai_platform",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-1-r2v",
"output_cost_per_second": 0.165026,
"output_cost_per_second_480p": 0.0618845,
"output_cost_per_second_720p": 0.123769,
"output_cost_per_second_1080p": 0.165026,
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "reference_image"
},
"metadata": {
"comment": "HappyHorse 1.1 reference-to-video, 1 to 9 reference images. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwen_ai_platform/happyhorse-1.0-t2v": {
"litellm_provider": "qwen_ai_platform",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-t2v",
"output_cost_per_second": 0.220034,
"output_cost_per_second_720p": 0.123769,
"output_cost_per_second_1080p": 0.220034,
"supported_modalities": [
"text"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "HappyHorse 1.0 text-to-video. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwen_ai_platform/happyhorse-1.0-i2v": {
"litellm_provider": "qwen_ai_platform",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-i2v",
"output_cost_per_second": 0.220034,
"output_cost_per_second_720p": 0.123769,
"output_cost_per_second_1080p": 0.220034,
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "HappyHorse 1.0 image-to-video from a first frame. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwen_ai_platform/happyhorse-1.0-r2v": {
"litellm_provider": "qwen_ai_platform",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/happyhorse-1-0-r2v",
"output_cost_per_second": 0.220034,
"output_cost_per_second_720p": 0.123769,
"output_cost_per_second_1080p": 0.220034,
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "reference_image"
},
"metadata": {
"comment": "HappyHorse 1.0 reference-to-video. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwen_ai_platform/wan2.7-t2v": {
"litellm_provider": "qwen_ai_platform",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-t2v",
"output_cost_per_second": 0.143353,
"output_cost_per_second_720p": 0.086012,
"output_cost_per_second_1080p": 0.143353,
"supported_modalities": [
"text",
"audio"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "Wan 2.7 text-to-video. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwen_ai_platform/wan2.7-i2v": {
"litellm_provider": "qwen_ai_platform",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-i2v",
"output_cost_per_second": 0.143353,
"output_cost_per_second_720p": 0.086012,
"output_cost_per_second_1080p": 0.143353,
"supported_modalities": [
"text",
"image",
"video",
"audio"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "first_frame"
},
"metadata": {
"comment": "Wan 2.7 image-to-video (first frame, first and last frame, continuation). No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"qwen_ai_platform/wan2.7-r2v": {
"litellm_provider": "qwen_ai_platform",
"mode": "video_generation",
"source": "https://www.alibabacloud.com/help/en/model-studio/wan2-7-r2v",
"output_cost_per_second": 0.143353,
"output_cost_per_second_720p": 0.086012,
"output_cost_per_second_1080p": 0.143353,
"supported_modalities": [
"text",
"image",
"video",
"audio"
],
"supported_output_modalities": [
"video"
],
"supported_endpoints": [
"/v1/videos"
],
"provider_specific_entry": {
"dashscope_video_reference_input": "reference_image"
},
"metadata": {
"comment": "Wan 2.7 reference-to-video. No 480P tier. China (Beijing) list price per output second, USD, matching this prefix's default host. The base rate is the 1080P default tier."
}
},
"databricks/databricks-bge-large-en": {
"cache_creation_input_token_cost": 1.0003e-07,
"cache_read_input_token_cost": 1.0003e-07,

View file

@ -0,0 +1,818 @@
"""
Tests for DashScope (Wan 3.0 / Wan 2.7 / HappyHorse) video generation transformation.
"""
import base64
import io
import json
from datetime import datetime, timedelta, timezone
from typing import Final
from unittest.mock import Mock
import httpx
import pytest
from litellm.exceptions import UnsupportedParamsError
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler
from litellm.llms.dashscope.qwen_ai_platform import QwenAIPlatformVideoConfig
from litellm.llms.dashscope.qwencloud import QwenCloudVideoConfig
from litellm.llms.dashscope.videos.transformation import (
DashScopeVideoConfig,
DashScopeVideoError,
_parse_task_response,
_video_url_from_task,
)
from litellm.types.router import GenericLiteLLMParams
from litellm.types.videos.utils import (
decode_video_id_with_provider,
encode_video_id_with_provider,
)
from litellm.videos.main import avideo_generation
PNG_BYTES = base64.b64decode(
"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg=="
)
API_BASE = "https://dashscope.aliyuncs.com"
SYNTHESIS_URL = f"{API_BASE}/api/v1/services/aigc/video-generation/video-synthesis"
def _mock_response(payload: dict, status_code: int = 200) -> Mock:
mock_response = Mock(spec=httpx.Response)
mock_response.json.return_value = payload
mock_response.status_code = status_code
mock_response.headers = httpx.Headers()
return mock_response
def _create(params: dict, model: str = "wan3.0-video", prompt: str = "a cat on a roof"):
return DashScopeVideoConfig().transform_video_create_request(
model=model,
prompt=prompt,
api_base=API_BASE,
video_create_optional_request_params=params,
litellm_params=GenericLiteLLMParams(),
headers={},
)
class TestDashScopeVideoCreateRequest:
def test_body_is_nested_into_input_and_parameters(self):
"""DashScope rejects a flat Sora-shaped body: prompt and media belong
under ``input``, everything else under ``parameters``."""
data, files, url = _create({"resolution": "720P", "duration": 10, "ratio": "16:9"})
assert url == SYNTHESIS_URL
assert files == ()
assert data == {
"model": "wan3.0-video",
"input": {"prompt": "a cat on a roof"},
"parameters": {"resolution": "720P", "ratio": "16:9", "duration": 10},
}
def test_prompt_only_request_omits_parameters_block(self):
"""With nothing to configure, DashScope's documented text-to-video body
is just model + input, so no empty parameters block is sent."""
data, _, _ = _create({})
assert data == {"model": "wan3.0-video", "input": {"prompt": "a cat on a roof"}}
def test_media_array_passes_through_into_input(self):
media = [
{"type": "reference_image", "url": "https://x/1.png"},
{"type": "reference_audio", "url": "https://x/1.mp3"},
]
data, _, _ = _create({"media": media, "duration": 5})
assert data["input"]["media"] == media
assert "media" not in data["parameters"]
def test_negative_prompt_goes_to_input_not_parameters(self):
data, _, _ = _create({"negative_prompt": "blurry", "seed": 7})
assert data["input"]["negative_prompt"] == "blurry"
assert data["parameters"] == {"seed": 7}
class TestDashScopeVideoInputReference:
def test_wan3_file_reference_becomes_first_frame_media_data_uri(self):
"""Wan 3.0 takes references as typed media entries; a file must arrive
base64-encoded as a data URI rather than as an unusable file object."""
data, _, _ = _create({"input_reference": io.BytesIO(PNG_BYTES)})
media = data["input"]["media"]
assert len(media) == 1
assert media[0]["type"] == "first_frame"
assert media[0]["url"].startswith("data:image/png;base64,")
assert base64.b64decode(media[0]["url"].split(",", 1)[1]) == PNG_BYTES
def test_wan3_url_reference_is_passed_through_unencoded(self):
data, _, _ = _create({"input_reference": "https://x/first.png"})
assert data["input"]["media"] == ({"type": "first_frame", "url": "https://x/first.png"},)
@pytest.mark.parametrize(
"model", ["wan2.7-i2v", "wan2.7-i2v-2026-04-25", "happyhorse-1.1-i2v", "happyhorse-1.0-i2v"]
)
def test_current_i2v_models_take_the_reference_as_a_first_frame_media_entry(self, model):
"""Wan 2.7 and HappyHorse image-to-video take the typed media array;
a flat img_url is only understood by the legacy Wan 2.6 models."""
data, _, _ = _create({"input_reference": "https://x/first.png"}, model=model)
assert data["input"]["media"] == ({"type": "first_frame", "url": "https://x/first.png"},)
assert "img_url" not in data["input"]
@pytest.mark.parametrize("model", ["happyhorse-1.1-r2v", "happyhorse-1.0-r2v", "wan2.7-r2v"])
def test_reference_to_video_models_take_the_reference_as_a_reference_image(self, model):
"""The r2v models reject first_frame and only accept reference_image."""
data, _, _ = _create({"input_reference": "https://x/subject.png"}, model=model)
assert data["input"]["media"] == ({"type": "reference_image", "url": "https://x/subject.png"},)
@pytest.mark.parametrize(
"model", ["wan2.6-i2v-flash", "wan2.5-i2v-preview", "wan2.2-i2v-plus", "wanx2.1-i2v-turbo"]
)
def test_legacy_models_take_the_reference_on_the_flat_img_url_field(self, model):
"""Wan 2.6 and earlier predate the media array and would reject it."""
data, _, _ = _create({"input_reference": "https://x/first.png"}, model=model)
assert data["input"]["img_url"] == "https://x/first.png"
assert "media" not in data["input"]
def test_explicit_media_array_wins_over_input_reference(self):
"""Only the full media array can express multi-role references, so a
caller-supplied one must not be clobbered by input_reference."""
media = [{"type": "reference_video", "url": "https://x/v.mp4"}]
data, _, _ = _create({"media": media, "input_reference": "https://x/first.png"})
assert data["input"]["media"] == media
def test_legacy_explicit_img_url_wins_over_input_reference(self):
data, _, _ = _create(
{"img_url": "https://x/explicit.png", "input_reference": "https://x/other.png"},
model="wan2.6-i2v-flash",
)
assert data["input"]["img_url"] == "https://x/explicit.png"
class TestDashScopeVideoMapOpenAIParams:
def test_seconds_becomes_duration(self):
mapped = DashScopeVideoConfig().map_openai_params(
video_create_optional_params={"seconds": "10"}, model="wan3.0-video", drop_params=False
)
assert mapped == {"duration": 10}
assert "seconds" not in mapped
def test_size_becomes_both_ratio_and_resolution_tier(self):
"""DashScope takes a ratio plus a named resolution tier, and bills per
second by that tier, so a pixel size must produce both or the request
silently falls back to the pricier 1080P default."""
mapped = DashScopeVideoConfig().map_openai_params(
video_create_optional_params={"size": "1280x720"}, model="wan3.0-video", drop_params=False
)
assert mapped == {"ratio": "16:9", "resolution": "720P"}
@pytest.mark.parametrize(
"size,expected_resolution",
[
("854x480", "480P"),
("1280x720", "720P"),
("1920x1080", "1080P"),
("1080x1920", "1080P"),
("3840x2160", "1080P"),
],
)
def test_resolution_tier_is_picked_from_the_shortest_side(self, size, expected_resolution):
mapped = DashScopeVideoConfig().map_openai_params(
video_create_optional_params={"size": size}, model="wan3.0-video", drop_params=False
)
assert mapped["resolution"] == expected_resolution
def test_explicit_resolution_and_ratio_win_over_size(self):
mapped = DashScopeVideoConfig().map_openai_params(
video_create_optional_params={"size": "1280x720", "ratio": "4:3", "resolution": "1080P"},
model="wan3.0-video",
drop_params=False,
)
assert mapped["ratio"] == "4:3"
assert mapped["resolution"] == "1080P"
def test_openai_only_fields_are_dropped(self):
mapped = DashScopeVideoConfig().map_openai_params(
video_create_optional_params={"user": "u1", "characters": [{"id": "c"}], "prompt": "p"},
model="wan3.0-video",
drop_params=False,
)
assert mapped == {}
def test_parameters_block_is_merged(self):
mapped = DashScopeVideoConfig().map_openai_params(
video_create_optional_params={"parameters": {"prompt_extend": True, "audio": False}},
model="wan3.0-video",
drop_params=False,
)
assert mapped == {"prompt_extend": True, "audio": False}
def test_parameters_block_must_be_an_object(self):
with pytest.raises(ValueError, match="parameters must be an object"):
DashScopeVideoConfig().map_openai_params(
video_create_optional_params={"parameters": ["not", "an", "object"]},
model="wan3.0-video",
drop_params=False,
)
@pytest.mark.parametrize("params", [{"seconds": "-1"}, {"parameters": {"duration": -1}}])
def test_smart_duration_is_rejected_because_it_cannot_be_billed(self, params):
"""duration -1 lets DashScope pick the length after the create call,
which is the only call litellm bills, so it would record $0 for a
video DashScope charges for."""
mapped = DashScopeVideoConfig().map_openai_params(
video_create_optional_params=params, model="wan3.0-video", drop_params=False
)
with pytest.raises(UnsupportedParamsError, match="explicit duration") as raised:
_create(mapped)
assert raised.value.status_code == 400
class TestDashScopeVideoCreateResponse:
def test_task_id_is_encoded_with_provider_and_model(self):
"""The create response only carries task_id, so litellm must wrap it for
later status/content calls to route back to dashscope."""
video_obj = DashScopeVideoConfig().transform_video_create_response(
model="wan3.0-video",
raw_response=_mock_response(
{"output": {"task_status": "PENDING", "task_id": "0385dc79-5ff8"}, "request_id": "r1"}
),
logging_obj=None,
custom_llm_provider="dashscope",
request_data={"model": "wan3.0-video", "parameters": {"duration": 10, "resolution": "720P"}},
)
assert video_obj.status == "queued"
assert video_obj.model == "wan3.0-video"
decoded = decode_video_id_with_provider(video_obj.id)
assert decoded["custom_llm_provider"] == "dashscope"
assert decoded["model_id"] == "wan3.0-video"
assert decoded["video_id"] == "0385dc79-5ff8"
def test_usage_carries_requested_cost_inputs_while_queued(self):
video_obj = DashScopeVideoConfig().transform_video_create_response(
model="wan3.0-video",
raw_response=_mock_response({"output": {"task_status": "PENDING", "task_id": "t1"}}),
logging_obj=None,
custom_llm_provider="dashscope",
request_data={"parameters": {"duration": 4, "resolution": "480P"}},
)
assert video_obj.usage == {"duration_seconds": 4.0, "video_resolution": "480p"}
@pytest.mark.parametrize(
"parameters,expected_usage",
[
(None, {"duration_seconds": 5.0, "video_resolution": "1080p"}),
({"resolution": "480P"}, {"duration_seconds": 5.0, "video_resolution": "480p"}),
({"duration": 8}, {"duration_seconds": 8.0, "video_resolution": "1080p"}),
],
)
def test_omitted_duration_and_tier_bill_dashscope_defaults(self, parameters, expected_usage):
"""DashScope fills an omitted duration (5s) and resolution (1080P) and
bills them; the create call is the billed one, so leaving either out
recorded the video at $0."""
video_obj = DashScopeVideoConfig().transform_video_create_response(
model="wan3.0-video",
raw_response=_mock_response({"output": {"task_status": "PENDING", "task_id": "t1"}}),
logging_obj=None,
custom_llm_provider="dashscope",
request_data={"model": "wan3.0-video", **({"parameters": parameters} if parameters else {})},
)
assert video_obj.usage == expected_usage
def test_in_body_error_on_a_200_is_raised(self):
"""DashScope reports create failures as a 200 with top-level code and no
output; without this the caller gets a queued video with an empty id."""
with pytest.raises(DashScopeVideoError, match="No API-key provided"):
DashScopeVideoConfig().transform_video_create_response(
model="wan3.0-video",
raw_response=_mock_response(
{"code": "InvalidApiKey", "message": "No API-key provided.", "request_id": "r1"}
),
logging_obj=None,
custom_llm_provider="dashscope",
request_data={},
)
class TestDashScopeVideoStatus:
SUCCEEDED = {
"request_id": "78c9b768",
"output": {
"task_id": "17ed7e50",
"task_status": "SUCCEEDED",
"submit_time": "2026-08-06 10:01:35.452",
"scheduled_time": "2026-08-06 10:01:35.507",
"end_time": "2026-08-06 10:13:33.838",
"orig_prompt": "a golden retriever",
"video_url": "https://oss.example.com/video.mp4",
},
"usage": {
"video_count": 1,
"duration": 5.0,
"input_video_duration": 0.0,
"output_video_duration": 5.0,
"fps": 30,
"SR": 720,
"ratio": "16:9",
},
}
def test_succeeded_task_mapping(self):
video_obj = DashScopeVideoConfig().transform_video_status_retrieve_response(
raw_response=_mock_response(self.SUCCEEDED),
logging_obj=None,
custom_llm_provider="dashscope",
)
assert video_obj.status == "completed"
assert video_obj.seconds == "5.0"
assert video_obj.completed_at is not None
assert video_obj.created_at is not None
assert video_obj.completed_at > video_obj.created_at
assert decode_video_id_with_provider(video_obj.id)["video_id"] == "17ed7e50"
def test_timestamps_are_read_as_utc_plus_8(self):
"""DashScope stamps times in UTC+8 with no offset; reading them as UTC
would shift every video's created_at by 8 hours."""
video_obj = DashScopeVideoConfig().transform_video_status_retrieve_response(
raw_response=_mock_response(self.SUCCEEDED), logging_obj=None, custom_llm_provider="dashscope"
)
submitted = datetime(2026, 8, 6, 10, 1, 35, 452000, tzinfo=timezone(timedelta(hours=8)))
assert video_obj.created_at == int(submitted.timestamp())
def test_delivered_resolution_beats_the_requested_one_for_billing(self):
"""ratio adaptive lets DashScope deliver a different tier than asked
for; usage.SR is what was produced and therefore what is billed."""
video_obj = DashScopeVideoConfig().transform_video_status_retrieve_response(
raw_response=_mock_response(self.SUCCEEDED), logging_obj=None, custom_llm_provider="dashscope"
)
assert video_obj.usage == {"duration_seconds": 5.0, "video_resolution": "720p"}
def test_billed_duration_includes_input_video_seconds(self):
"""For edit and extend, DashScope bills usage.duration (output plus
input video seconds), which is larger than the output alone."""
payload = {
"output": {"task_id": "t1", "task_status": "SUCCEEDED", "video_url": "https://x/v.mp4"},
"usage": {"duration": 12.0, "input_video_duration": 7.0, "output_video_duration": 5.0, "SR": 1080},
}
video_obj = DashScopeVideoConfig().transform_video_status_retrieve_response(
raw_response=_mock_response(payload), logging_obj=None, custom_llm_provider="dashscope"
)
assert video_obj.usage["duration_seconds"] == 12.0
assert video_obj.usage["video_resolution"] == "1080p"
def test_size_is_rebuilt_from_shortest_side_and_ratio(self):
"""DashScope splits geometry across SR and ratio; OpenAI's size field is
pixels, so reporting the bare ratio would be meaningless."""
video_obj = DashScopeVideoConfig().transform_video_status_retrieve_response(
raw_response=_mock_response(self.SUCCEEDED), logging_obj=None, custom_llm_provider="dashscope"
)
assert video_obj.size == "1280x720"
@pytest.mark.parametrize(
"task_status,expected",
[
("PENDING", "queued"),
("RUNNING", "in_progress"),
("SUCCEEDED", "completed"),
("FAILED", "failed"),
("CANCELED", "cancelled"),
("UNKNOWN", "failed"),
],
)
def test_status_mapping(self, task_status, expected):
video_obj = DashScopeVideoConfig().transform_video_status_retrieve_response(
raw_response=_mock_response({"output": {"task_id": "t1", "task_status": task_status}}),
logging_obj=None,
custom_llm_provider="dashscope",
)
assert video_obj.status == expected
def test_running_task_has_no_completion_time(self):
video_obj = DashScopeVideoConfig().transform_video_status_retrieve_response(
raw_response=_mock_response(
{"output": {"task_id": "t1", "task_status": "RUNNING", "submit_time": "2026-08-06 10:01:35.452"}}
),
logging_obj=None,
custom_llm_provider="dashscope",
)
assert video_obj.completed_at is None
assert video_obj.usage is None
def test_failed_task_maps_error_from_inside_output(self):
"""Unlike creation, task failures nest code and message inside output."""
video_obj = DashScopeVideoConfig().transform_video_status_retrieve_response(
raw_response=_mock_response(
{
"output": {
"task_id": "eff1443c",
"task_status": "FAILED",
"code": "InvalidParameter",
"message": "The two modes are mutually exclusive.",
}
}
),
logging_obj=None,
custom_llm_provider="dashscope",
)
assert video_obj.status == "failed"
assert video_obj.error == {
"code": "InvalidParameter",
"message": "The two modes are mutually exclusive.",
}
def test_polled_id_keeps_the_deployment_it_was_polled_with(self):
"""The proxy resolves the deployment from the model encoded in the id;
dropping it from the returned id sent follow-up polls and downloads to
the default provider credentials instead of the deployment's."""
polled_id = encode_video_id_with_provider("17ed7e50", "dashscope", "wan3-prod-deployment")
logging_obj = Mock(litellm_params={"video_id": polled_id})
video_obj = DashScopeVideoConfig().transform_video_status_retrieve_response(
raw_response=_mock_response(self.SUCCEEDED), logging_obj=logging_obj, custom_llm_provider="dashscope"
)
decoded = decode_video_id_with_provider(video_obj.id)
assert decoded["model_id"] == "wan3-prod-deployment"
assert decoded["video_id"] == "17ed7e50"
assert video_obj.model == "wan3-prod-deployment"
def test_in_body_lookup_error_is_raised_instead_of_a_queued_video(self):
"""An unknown or expired task id comes back as a 200 with top-level code
and no output; mapping it as a task produced a queued video with an
empty id that callers would poll forever."""
with pytest.raises(DashScopeVideoError, match="task not found"):
DashScopeVideoConfig().transform_video_status_retrieve_response(
raw_response=_mock_response({"code": "InvalidParameter", "message": "task not found"}),
logging_obj=None,
custom_llm_provider="dashscope",
)
def test_request_decodes_task_id_from_wrapped_video_id(self):
encoded = encode_video_id_with_provider("17ed7e50", "dashscope", "wan3.0-video")
url, params = DashScopeVideoConfig().transform_video_status_retrieve_request(
video_id=encoded, api_base=API_BASE, litellm_params=GenericLiteLLMParams(), headers={}
)
assert url == f"{API_BASE}/api/v1/tasks/17ed7e50"
assert params == {}
class TestDashScopeVideoContent:
def test_content_request_targets_the_task_endpoint(self):
encoded = encode_video_id_with_provider("17ed7e50", "dashscope", "wan3.0-video")
url, params = DashScopeVideoConfig().transform_video_content_request(
video_id=encoded, api_base=API_BASE, litellm_params=GenericLiteLLMParams(), headers={}
)
assert url == f"{API_BASE}/api/v1/tasks/17ed7e50"
assert params == {}
def test_url_is_extracted_from_a_succeeded_task(self):
task = _parse_task_response(
_mock_response(
{"output": {"task_id": "t1", "task_status": "SUCCEEDED", "video_url": "https://oss/video.mp4"}}
)
)
assert _video_url_from_task(task) == "https://oss/video.mp4"
@pytest.mark.parametrize("task_status", ["PENDING", "RUNNING"])
def test_pending_task_raises_still_processing(self, task_status):
task = _parse_task_response(_mock_response({"output": {"task_id": "t1", "task_status": task_status}}))
with pytest.raises(ValueError, match="still processing"):
_video_url_from_task(task)
def test_failed_task_surfaces_the_upstream_message(self):
task = _parse_task_response(
_mock_response(
{"output": {"task_id": "t1", "task_status": "FAILED", "message": "content policy violation"}}
)
)
with pytest.raises(ValueError, match="content policy violation"):
_video_url_from_task(task)
def test_expired_task_explains_the_24h_window(self):
"""A task id older than 24h comes back UNKNOWN with no error, which is
otherwise indistinguishable from a bad id."""
task = _parse_task_response(_mock_response({"output": {"task_id": "t1", "task_status": "UNKNOWN"}}))
with pytest.raises(ValueError, match="24 hours"):
_video_url_from_task(task)
class TestDashScopeVideoEnvironment:
def test_async_header_is_always_sent(self):
"""DashScope rejects video-synthesis calls without X-DashScope-Async,
with 'current user api does not support synchronous calls'."""
headers = DashScopeVideoConfig().validate_environment(
headers={}, model="wan3.0-video", api_key="sk-test", litellm_params=GenericLiteLLMParams()
)
assert headers["X-DashScope-Async"] == "enable"
assert headers["Authorization"] == "Bearer sk-test"
def test_api_key_from_litellm_params(self):
headers = DashScopeVideoConfig().validate_environment(
headers={}, model="wan3.0-video", api_key=None, litellm_params=GenericLiteLLMParams(api_key="sk-params")
)
assert headers["Authorization"] == "Bearer sk-params"
def test_missing_api_key_raises(self, monkeypatch):
monkeypatch.delenv("DASHSCOPE_API_KEY", raising=False)
monkeypatch.setattr("litellm.api_key", None)
with pytest.raises(ValueError, match="DASHSCOPE_API_KEY"):
DashScopeVideoConfig().validate_environment(
headers={}, model="wan3.0-video", api_key=None, litellm_params=GenericLiteLLMParams()
)
def test_compatible_mode_chat_base_is_stripped_back_to_the_host(self):
"""DASHSCOPE_API_BASE is shared with chat and points at
/compatible-mode/v1; the video API lives under /api/v1."""
url = DashScopeVideoConfig().get_complete_url(
model="wan3.0-video",
api_base="https://dashscope.aliyuncs.com/compatible-mode/v1",
litellm_params={},
)
assert url == API_BASE
def test_workspace_scoped_host_is_preserved(self):
url = DashScopeVideoConfig().get_complete_url(
model="wan3.0-video", api_base="https://ws-123.cn-beijing.maas.aliyuncs.com/", litellm_params={}
)
assert url == "https://ws-123.cn-beijing.maas.aliyuncs.com"
def test_default_host_is_beijing(self, monkeypatch):
monkeypatch.delenv("DASHSCOPE_API_BASE_VIDEO", raising=False)
assert (
DashScopeVideoConfig().get_complete_url(model="wan3.0-video", api_base=None, litellm_params={}) == API_BASE
)
class TestDashScopeVideoBrandAliases:
def test_qwencloud_defaults_to_the_international_host(self, monkeypatch):
monkeypatch.delenv("QWENCLOUD_API_BASE_VIDEO", raising=False)
assert (
QwenCloudVideoConfig().get_complete_url(model="wan3.0-video", api_base=None, litellm_params={})
== "https://dashscope-intl.aliyuncs.com"
)
def test_qwen_ai_platform_defaults_to_the_china_host(self, monkeypatch):
monkeypatch.delenv("QWEN_AI_PLATFORM_API_BASE_VIDEO", raising=False)
assert (
QwenAIPlatformVideoConfig().get_complete_url(model="wan3.0-video", api_base=None, litellm_params={})
== API_BASE
)
def test_brand_alias_falls_back_to_the_shared_dashscope_key(self, monkeypatch):
monkeypatch.delenv("QWENCLOUD_API_KEY", raising=False)
monkeypatch.setenv("DASHSCOPE_API_KEY", "sk-shared")
monkeypatch.setattr("litellm.api_key", None)
headers = QwenCloudVideoConfig().validate_environment(
headers={}, model="wan3.0-video", api_key=None, litellm_params=GenericLiteLLMParams()
)
assert headers["Authorization"] == "Bearer sk-shared"
class TestDashScopeVideoUnsupportedOperations:
@pytest.mark.parametrize("operation", ["remix", "list", "delete"])
def test_operations_dashscope_does_not_expose_raise(self, operation):
"""DashScope publishes no remix, list or delete surface, so these must
fail loudly instead of silently hitting a made-up endpoint."""
config = DashScopeVideoConfig()
calls = {
"remix": lambda: config.transform_video_remix_request(
video_id="v", prompt="p", api_base=API_BASE, litellm_params=GenericLiteLLMParams(), headers={}
),
"list": lambda: config.transform_video_list_request(
api_base=API_BASE, litellm_params=GenericLiteLLMParams(), headers={}
),
"delete": lambda: config.transform_video_delete_request(
video_id="v", api_base=API_BASE, litellm_params=GenericLiteLLMParams(), headers={}
),
}
with pytest.raises(NotImplementedError, match="not supported by DashScope"):
calls[operation]()
class TestDashScopeVideoProviderWiring:
@pytest.mark.parametrize(
"provider,expected",
[
("dashscope", DashScopeVideoConfig),
("qwencloud", QwenCloudVideoConfig),
("qwen_ai_platform", QwenAIPlatformVideoConfig),
],
)
def test_provider_resolves_to_a_video_config(self, provider, expected):
"""Without this registration litellm answers 'video generation is not
supported for dashscope' before any transform runs."""
import litellm
from litellm.utils import ProviderConfigManager
config = ProviderConfigManager.get_provider_video_config(
model="wan3.0-video", provider=litellm.LlmProviders(provider)
)
assert type(config) is expected
VIDEO_MODEL_TIERS: Final = (
("wan3.0-video", ("480p", "720p", "1080p")),
("wan3.0-video-prime", ("480p", "720p", "1080p")),
("happyhorse-1.1-t2v", ("480p", "720p", "1080p")),
("happyhorse-1.1-i2v", ("480p", "720p", "1080p")),
("happyhorse-1.1-r2v", ("480p", "720p", "1080p")),
("happyhorse-1.0-t2v", ("720p", "1080p")),
("happyhorse-1.0-i2v", ("720p", "1080p")),
("happyhorse-1.0-r2v", ("720p", "1080p")),
("wan2.7-t2v", ("720p", "1080p")),
("wan2.7-i2v", ("720p", "1080p")),
("wan2.7-r2v", ("720p", "1080p")),
)
@pytest.mark.usefixtures("local_model_cost_map")
class TestDashScopeVideoPricing:
@pytest.mark.parametrize("provider", ["dashscope", "qwencloud", "qwen_ai_platform"])
@pytest.mark.parametrize("model,tiers", VIDEO_MODEL_TIERS)
def test_every_supported_tier_bills_its_own_rate(self, provider, model, tiers):
"""A tier without its own rate falls back to the base rate, so a 480P
video would silently bill at the 1080P price; a missing alias entry
bills the whole video at zero."""
from litellm import get_model_info
from litellm.llms.openai.cost_calculation import video_generation_cost
info = get_model_info(model=model, custom_llm_provider=provider)
for tier in tiers:
tier_rate = info[f"output_cost_per_second_{tier}"]
assert tier_rate > 0
cost = video_generation_cost(
model=model, duration_seconds=5.0, custom_llm_provider=provider, video_resolution=tier
)
assert cost == pytest.approx(tier_rate * 5.0)
@pytest.mark.parametrize("provider", ["dashscope", "qwencloud", "qwen_ai_platform"])
@pytest.mark.parametrize("model,tiers", VIDEO_MODEL_TIERS)
def test_higher_tiers_never_bill_less_and_1080p_is_the_default(self, provider, model, tiers):
"""DashScope defaults to 1080P, so an unlabelled video must bill at the
1080P rate rather than the cheapest tier."""
from litellm import get_model_info
info = get_model_info(model=model, custom_llm_provider=provider)
rates = [info[f"output_cost_per_second_{tier}"] for tier in tiers]
assert rates == sorted(rates)
assert info["output_cost_per_second"] == info["output_cost_per_second_1080p"]
@pytest.mark.parametrize("model,_", VIDEO_MODEL_TIERS)
def test_each_prefix_is_priced_for_the_region_its_default_host_serves(self, model, _):
"""dashscope and qwen_ai_platform default to the Beijing host and
qwencloud to the international one, so the two Beijing-routed prefixes
must share a rate card."""
from litellm import get_model_info
dashscope = get_model_info(model=model, custom_llm_provider="dashscope")
qwen_ai_platform = get_model_info(model=model, custom_llm_provider="qwen_ai_platform")
assert dashscope["output_cost_per_second_1080p"] == qwen_ai_platform["output_cost_per_second_1080p"]
class TestDashScopeVideoEndToEndRequest:
"""
Drive the real litellm entrypoint rather than calling the transform directly.
``video_generation()`` runs map_openai_params first and passes only the
*mapped* params to transform_video_create_request, so anything the mapper
drops never reaches the request builder. Asserting on the JSON the handler
actually serialized also catches body fragments that aren't JSON-encodable.
"""
@staticmethod
async def _wire_request(**kwargs) -> dict:
sent: Final[list[httpx.Request]] = []
def respond(request: httpx.Request) -> httpx.Response:
sent.append(request)
return httpx.Response(200, json={"output": {"task_status": "PENDING", "task_id": "t1"}})
async with httpx.AsyncClient(transport=httpx.MockTransport(respond)) as http_client:
handler: Final = AsyncHTTPHandler()
await handler.close()
handler.client = http_client
await avideo_generation(api_key="sk-test", api_base=API_BASE, client=handler, **kwargs)
(request,) = sent
return {"serialized": json.loads(request.content), "headers": dict(request.headers)}
@pytest.mark.asyncio
async def test_input_reference_survives_param_mapping_into_the_body(self):
"""Regression: map_openai_params used to drop input_reference, so an
image-to-video call silently degraded to text-to-video."""
captured = await self._wire_request(
model="dashscope/wan3.0-video",
prompt="make it move",
input_reference="https://x/first.png",
seconds="10",
)
assert captured["serialized"]["input"]["media"] == [{"type": "first_frame", "url": "https://x/first.png"}]
assert captured["serialized"]["parameters"]["duration"] == 10
assert "input_reference" not in captured["serialized"]["input"]
assert "input_reference" not in captured["serialized"].get("parameters", {})
@pytest.mark.asyncio
async def test_happyhorse_r2v_input_reference_reaches_the_body_as_a_reference_image(self):
captured = await self._wire_request(
model="dashscope/happyhorse-1.1-r2v",
prompt="make it move",
input_reference="https://x/subject.png",
)
assert captured["serialized"]["input"]["media"] == [{"type": "reference_image", "url": "https://x/subject.png"}]
@pytest.mark.asyncio
async def test_file_input_reference_is_json_encodable(self):
"""Regression: the first_frame entry was built as a MappingProxyType,
which json.dumps refuses, so every i2v request raised before sending."""
captured = await self._wire_request(
model="dashscope/wan3.0-video",
prompt="make it move",
input_reference=io.BytesIO(PNG_BYTES),
)
media = captured["serialized"]["input"]["media"]
assert media[0]["type"] == "first_frame"
assert media[0]["url"].startswith("data:image/png;base64,")
@pytest.mark.asyncio
async def test_auth_and_async_headers_reach_the_wire(self):
"""DashScope rejects the call outright without X-DashScope-Async."""
captured = await self._wire_request(model="dashscope/wan3.0-video", prompt="p")
headers = {k.lower(): v for k, v in captured["headers"].items()}
assert headers["x-dashscope-async"] == "enable"
assert headers["authorization"] == "Bearer sk-test"
@pytest.mark.asyncio
async def test_size_reaches_the_body_as_ratio_plus_resolution(self):
captured = await self._wire_request(
model="dashscope/happyhorse-1.1-t2v", prompt="a train", seconds="5", size="1920x1080"
)
assert captured["serialized"]["parameters"] == {
"resolution": "1080P",
"ratio": "16:9",
"duration": 5,
}
@pytest.mark.asyncio
async def test_multi_role_media_array_reaches_the_body_intact(self):
media = [
{"type": "reference_video", "url": "https://x/v.mp4"},
{"type": "reference_audio", "url": "https://x/a.mp3"},
]
captured = await self._wire_request(
model="dashscope/wan3.0-video", prompt="视频1", media=media, duration=15, resolution="1080P"
)
assert captured["serialized"]["input"]["media"] == media
assert captured["serialized"]["parameters"]["duration"] == 15