From 1f8ae05bc4e4c84cf491f0ef3cddfcd67307fcf3 Mon Sep 17 00:00:00 2001 From: Rick <26716961+Bytechoreographer@users.noreply.github.com> Date: Tue, 29 Sep 2026 18:18:36 +0800 Subject: [PATCH] feat(minimax): add MiniMax-H3 and H3-Max video generation via /v1/videos MinimaxVideoConfig maps litellm's OpenAI Videos API onto MiniMax's V2 task API. Create, status, list, content and delete hit /v2/video_generation and /v2/query/video_generation. Remix is not exposed: MiniMax's /v2/video_regeneration only upscales a finished 768P task to 2K and ignores any prompt, so it raises instead of returning a video that silently drops the edit OpenAI params translate to MiniMax's contract: seconds becomes duration, size reduces via gcd to a valid ratio, and text-to-video requests default to ratio 16:9, resolution 768P and duration 5 when unset. input_reference becomes a first_frame content item, base64 encoded into a data URI for file inputs, and multimodal reference media passes through a content array in extra_body. Caller content is validated strictly as image, video or audio items, and the scanned prompt is always the one text item sent, so text can't reach MiniMax past the guardrails; anything else in content returns a 400. api_base shares MINIMAX_API_BASE with the chat config (a trailing /v1 or /v2 is stripped because the video API carries its own version). The list endpoint pages by number with no cursor or sort, so after and an ascending order are rejected with a 400 rather than ignored, and page_num passes through extra_query. Responses validate into frozen pydantic models at the boundary, video ids are provider-encoded so status and content route back to MiniMax, and create reports duration_seconds and video_resolution in usage for cost tracking End-to-end tests drive the real avideo_generation() entrypoint through an httpx MockTransport and assert on the encoded request body, which is what catches a mapper that drops input_reference before the request is built Both model_prices maps gain minimax/MiniMax-H3 and minimax/MiniMax-H3-Max, priced per output second by resolution tier from MiniMax's pay-as-you-go list (H3 $0.08 at 768P and $0.13 at 2K, H3-Max $0.05 at 480P and $0.08 at 768P). MiniMax also bills input images beyond a per-model free allowance (5 for H3, 2 for H3-Max), so each entry carries input_cost_per_image and provider_specific_entry.minimax_free_input_images, and the create response reports the full charge as provider_reported_cost_usd. Reference video input is rejected with a 400, because MiniMax bills its duration only after the create call that litellm bills Co-authored-by: AaronHowell <237895480@qq.com> --- litellm/llms/minimax/videos/__init__.py | 0 litellm/llms/minimax/videos/transformation.py | 692 +++++++++++++++++ ...odel_prices_and_context_window_backup.json | 54 ++ litellm/utils.py | 4 + model_prices_and_context_window.json | 54 ++ tests/unit/llms/minimax/videos/__init__.py | 0 .../test_minimax_video_transformation.py | 697 ++++++++++++++++++ 7 files changed, 1501 insertions(+) create mode 100644 litellm/llms/minimax/videos/__init__.py create mode 100644 litellm/llms/minimax/videos/transformation.py create mode 100644 tests/unit/llms/minimax/videos/__init__.py create mode 100644 tests/unit/llms/minimax/videos/test_minimax_video_transformation.py diff --git a/litellm/llms/minimax/videos/__init__.py b/litellm/llms/minimax/videos/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/litellm/llms/minimax/videos/transformation.py b/litellm/llms/minimax/videos/transformation.py new file mode 100644 index 00000000000..5962213b1fe --- /dev/null +++ b/litellm/llms/minimax/videos/transformation.py @@ -0,0 +1,692 @@ +""" +MiniMax V2 video task API: create, poll, list, download and delete. Regeneration (/v2/video_regeneration) +only upscales a finished 768P task to 2K and ignores the prompt, so it is not exposed as an OpenAI remix. +""" + +import base64 +from collections.abc import Mapping, Sequence +from io import BufferedReader, BytesIO +from math import gcd +from types import MappingProxyType +from typing import TYPE_CHECKING, Final, Literal + +import httpx +from httpx._types import RequestFiles +from pydantic import BaseModel, TypeAdapter, ValidationError + +import litellm +from litellm.exceptions import UnsupportedParamsError +from litellm.images.utils import ImageEditRequestUtils +from litellm.litellm_core_utils.url_utils import encode_url_path_segment +from litellm.llms.base_llm.chat.transformation import BaseLLMException +from litellm.llms.base_llm.videos.transformation import BaseVideoConfig +from litellm.llms.custom_httpx.http_handler import ( + _get_httpx_client, # pyright: ignore[reportPrivateUsage, reportUnknownVariableType] # house cached-client factory has no public alias and its stub leaves params untyped + get_async_httpx_client, # pyright: ignore[reportUnknownVariableType] # factory stub leaves params untyped +) +from litellm.llms.openai.cost_calculation import video_generation_cost +from litellm.secret_managers.main import get_secret_str +from litellm.types.router import GenericLiteLLMParams +from litellm.types.videos.main import VideoCreateOptionalRequestParams, VideoObject +from litellm.types.videos.utils import ( + encode_video_id_with_provider, + extract_original_video_id, +) +from litellm.utils import get_model_info + +if TYPE_CHECKING: + from litellm.litellm_core_utils.litellm_logging import Logging + from litellm.llms.custom_httpx.http_handler import HTTPHandler + + +class MinimaxVideoError(BaseLLMException): + pass + + +class _TaskError(BaseModel, frozen=True): + code: str | None = None + message: str | None = None + + +class _TaskUsage(BaseModel, frozen=True): + total_seconds: int | None = None + input_seconds: int | None = None + output_seconds: int | None = None + input_image_count: int | None = None + input_audio_seconds: int | None = None + total_tokens: int | None = None + prompt_tokens: int | None = None + completion_tokens: int | None = None + + +class _TaskContent(BaseModel, frozen=True): + url: str | None = None + prompt: str | None = None + + +class _ContentItem(BaseModel, frozen=True): + type: str = "" + text: str | None = None + role: str | None = None + + +class _MediaUrl(BaseModel, frozen=True, extra="forbid"): + url: str + + +class _CallerMediaItem(BaseModel, frozen=True, extra="forbid"): + type: Literal["image_url", "video_url", "audio_url"] + image_url: _MediaUrl | None = None + video_url: _MediaUrl | None = None + audio_url: _MediaUrl | None = None + role: Literal["first_frame", "last_frame", "reference_image", "reference_video", "reference_audio"] | None = None + + +class _MiniMaxTask(BaseModel, frozen=True): + id: str = "" + model: str | None = None + status: str = "" + error: _TaskError | None = None + created_at: int | None = None + updated_at: int | None = None + content: _TaskContent | None = None + resolution: str | None = None + duration: int | None = None + usage: _TaskUsage | None = None + ratio: str | None = None + task_type: str | None = None + modality: str | None = None + + +class _TaskIdResponse(BaseModel, frozen=True): + task_id: str = "" + + +class _TaskResponse(BaseModel, frozen=True): + task: _MiniMaxTask | None = None + + +class _ListResponse(BaseModel, frozen=True): + items: Sequence[_MiniMaxTask] | None = None + total: int | None = None + + +class _DeleteResponse(BaseModel, frozen=True): + task_id: str = "" + status: str = "" + + +MINIMAX_VIDEO_DEFAULT_API_BASE: Final = "https://api.minimax.io" +MINIMAX_VIDEO_DEFAULT_RESOLUTION: Final = "768P" +MINIMAX_VIDEO_DEFAULT_DURATION_SECONDS: Final = 5 +MINIMAX_TEXT_TO_VIDEO_DEFAULT_RATIO: Final = "16:9" + +_EMPTY_PARAMS: Final[dict[str, object]] = {} # mutable-ok: BaseVideoConfig contract empty params + +_RESPONSE_ADAPTER: Final = TypeAdapter(dict[str, object]) +_TASK_ID_RESPONSE_ADAPTER: Final = TypeAdapter(_TaskIdResponse) +_TASK_RESPONSE_ADAPTER: Final = TypeAdapter(_TaskResponse) +_LIST_RESPONSE_ADAPTER: Final = TypeAdapter(_ListResponse) +_DELETE_RESPONSE_ADAPTER: Final = TypeAdapter(_DeleteResponse) +_CONTENT_ITEMS_ADAPTER: Final = TypeAdapter(list[_ContentItem]) +_CALLER_MEDIA_ADAPTER: Final = TypeAdapter(list[_CallerMediaItem]) + +_STATUS_MAP: Final = MappingProxyType( + { + "queued": "queued", + "running": "in_progress", + "succeeded": "completed", + "failed": "failed", + "cancelled": "cancelled", + } +) + +_DROP_FROM_CREATE_BODY: Final = frozenset( + { + "model", + "prompt", + "user", + "characters", + "image", + "extra_headers", + "extra_query", + "extra_body", + } +) + +_CREATE_BODY_KEYS: Final = ( + "resolution", + "duration", + "ratio", + "callback_url", + "aigc_watermark", + "extra", +) + + +def _parse_task_response(raw_response: httpx.Response) -> _MiniMaxTask: + return _TASK_RESPONSE_ADAPTER.validate_python(raw_response.json()).task or _MiniMaxTask() + + +def _parse_task_id_response(raw_response: httpx.Response) -> str: + return _TASK_ID_RESPONSE_ADAPTER.validate_python(raw_response.json()).task_id + + +def _parse_delete_response(raw_response: httpx.Response) -> tuple[str, str]: + deleted: Final = _DELETE_RESPONSE_ADAPTER.validate_python(raw_response.json()) + return deleted.task_id, deleted.status + + +MINIMAX_SURFACE_SUFFIXES: Final = ("/v1", "/v2", "/anthropic") + + +def _normalized_api_base(api_base: str) -> str: + """ + A MiniMax key covers the chat surfaces (``/v1``, ``/anthropic``) and video (``/v2``), so strip whichever + surface suffix a shared api_base carries back to the host. + """ + trimmed: Final = api_base.rstrip("/") + matched: Final = next((suffix for suffix in MINIMAX_SURFACE_SUFFIXES if trimmed.endswith(suffix)), None) + return trimmed[: -len(matched)] if matched else trimmed + + +def _ratio_from_size(size: str) -> str | None: + if ":" in size: + return size + width_str, separator, height_str = size.partition("x") + if not separator or not (width_str.isdigit() and height_str.isdigit()): + return None + width: Final = int(width_str) + height: Final = int(height_str) + divisor: Final = gcd(width, height) + return f"{width // divisor}:{height // divisor}" + + +def _duration_param(seconds: object) -> int | None: + if isinstance(seconds, bool): + return None + if isinstance(seconds, int): + return seconds + if not isinstance(seconds, str): + return None + try: + return int(float(seconds)) + except ValueError: + return None + + +def _read_all_bytes(file_obj: object) -> bytes: + if isinstance(file_obj, (BytesIO, BufferedReader)): + current_position: Final = file_obj.tell() + file_obj.seek(0) + content: Final = file_obj.read() + file_obj.seek(current_position) + return content + if isinstance(file_obj, bytes): + return file_obj + if isinstance(file_obj, bytearray): + return bytes(file_obj) + read: Final = getattr(file_obj, "read", None) + if callable(read): + data: Final = read() + if isinstance(data, bytes): + return data + raise ValueError("input_reference must be a URL string, bytes, or a file object") + + +def _image_url(image: object) -> str: + if isinstance(image, str): + return image + content_type: Final = ImageEditRequestUtils.get_image_content_type(image) + encoded: Final = base64.b64encode(_read_all_bytes(image)).decode("utf-8") + return f"data:{content_type};base64,{encoded}" + + +def _first_frame_content_item( + image: object, +) -> dict[str, object]: # mutable-ok: content items are JSON request-body fragments + return { # mutable-ok: request-body content item serialized to JSON by the handler + "type": "image_url", + "image_url": {"url": _image_url(image)}, # mutable-ok: request-body content item field + "role": "first_frame", + } + + +def _content_items(content: object) -> tuple[_ContentItem, ...]: + if not isinstance(content, Sequence) or isinstance(content, (str, bytes)): + return () + try: + return tuple(_CONTENT_ITEMS_ADAPTER.validate_python(content)) + except ValidationError: + return () + + +def _is_text_only_content(content: object) -> bool: + if not isinstance(content, Sequence) or isinstance(content, (str, bytes)) or not content: + return False + try: + items: Final = _CONTENT_ITEMS_ADAPTER.validate_python(content) + except ValidationError: + return False + return all(item.type == "text" for item in items) + + +def _video_object_from_task(task: _MiniMaxTask) -> VideoObject: + status: Final = _STATUS_MAP.get(task.status, "queued") + usage_dump: Final = task.usage.model_dump(exclude_none=True) if task.usage is not None else None + return VideoObject( + id=task.id, + object="video", + status=status, + created_at=task.created_at, + completed_at=task.updated_at if status == "completed" else None, + error=task.error.model_dump(exclude_none=True) if task.error is not None else None, + seconds=str(task.duration) if task.duration is not None else None, + model=task.model, + usage=usage_dump or None, + ) + + +def _create_cost_usd(model: str, duration: float, resolution: str | None, input_image_count: int) -> float | None: + """ + MiniMax bills input images beyond a per-model free allowance on top of output seconds, and the create + call is the only billed one, so the whole charge is reported for the cost calculator to use as is. + """ + try: + info: Final = get_model_info(model=model, custom_llm_provider="minimax") + except Exception: + return None + provider_specific: Final = info.get("provider_specific_entry") + free_images: Final = ( + provider_specific.get("minimax_free_input_images") if isinstance(provider_specific, Mapping) else None + ) + image_rate: Final = info.get("input_cost_per_image") + if not isinstance(free_images, (int, float)) or not isinstance(image_rate, (int, float)): + return None + output_cost: Final = video_generation_cost( + model=model, duration_seconds=duration, custom_llm_provider="minimax", video_resolution=resolution + ) + return output_cost + max(0, input_image_count - int(free_images)) * image_rate + + +def _video_url_from_task(task: _MiniMaxTask) -> str: + if task.content is not None and task.content.url: + return task.content.url + + if task.status in ("queued", "running"): + raise ValueError(f"Video is still processing (status: {task.status}). Please wait and try again.") + if task.error is not None: + raise ValueError(f"Video generation failed: {task.error.message or 'unknown error'}") + raise ValueError("Video URL not found in task response. The task may not have succeeded yet.") + + +class MinimaxVideoConfig(BaseVideoConfig): + def get_supported_openai_params(self, model: str) -> list[str]: # mutable-ok: BaseVideoConfig contract returns list + return [ # mutable-ok: BaseVideoConfig contract returns list + "model", + "prompt", + "input_reference", + "seconds", + "size", + "resolution", + "user", + "extra_headers", + "duration", + "ratio", + "content", + "callback_url", + "aigc_watermark", + "extra", + "parameters", + ] + + def map_openai_params( + self, + video_create_optional_params: VideoCreateOptionalRequestParams, + model: str, + drop_params: bool, + ) -> dict[str, object]: # mutable-ok: BaseVideoConfig contract + mapped_params: Final[dict[str, object]] = {} # mutable-ok: BaseVideoConfig contract; extra_body merges into it + for key, value in video_create_optional_params.items(): + if value is None or key in _DROP_FROM_CREATE_BODY: + continue + if key == "seconds": + duration = _duration_param(value) + if duration is not None: + mapped_params["duration"] = duration + elif key == "size": + ratio = _ratio_from_size(value) if isinstance(value, str) else None + if ratio is not None: + mapped_params.setdefault("ratio", ratio) + elif key == "parameters": + try: + mapped_params.update(_RESPONSE_ADAPTER.validate_python(value)) + except ValidationError as e: + raise ValueError("parameters must be an object of MiniMax request fields") from e + else: + mapped_params[key] = value + return mapped_params + + def validate_environment( + self, + headers: dict[str, str], # mutable-ok: BaseVideoConfig contract; handler expects a mutable headers dict + model: str, + api_key: str | None = None, + litellm_params: GenericLiteLLMParams | None = None, + ) -> dict[str, str]: # mutable-ok: BaseVideoConfig contract + resolved_api_key: Final = ( + api_key + or (litellm_params.api_key if litellm_params is not None and litellm_params.api_key else None) + or litellm.api_key + or get_secret_str("MINIMAX_API_KEY") + ) + + if resolved_api_key is None: + raise ValueError( + "MiniMax API key is required. Set MINIMAX_API_KEY environment variable or pass api_key parameter." + ) + + auth_headers: Final[dict[str, str]] = { # mutable-ok: httpx request headers are a mutable dict + "Authorization": f"Bearer {resolved_api_key}", + "Content-Type": "application/json", + } + headers.update(auth_headers) + return headers + + def get_complete_url( + self, + model: str, + api_base: str | None, + litellm_params: dict[str, object], # mutable-ok: BaseVideoConfig contract + ) -> str: + resolved_api_base: Final = api_base or get_secret_str("MINIMAX_API_BASE") or MINIMAX_VIDEO_DEFAULT_API_BASE + return _normalized_api_base(resolved_api_base) + + def transform_video_create_request( + self, + model: str, + prompt: str, + api_base: str, + video_create_optional_request_params: dict[str, object], # mutable-ok: BaseVideoConfig contract + litellm_params: GenericLiteLLMParams, + headers: dict[str, str], # mutable-ok: BaseVideoConfig contract + ) -> tuple[dict[str, object], RequestFiles, str]: # mutable-ok: BaseVideoConfig contract + content: Final = self._content_param(video_create_optional_request_params, prompt) + if any(item.type == "video_url" for item in _content_items(content)): + raise UnsupportedParamsError( + message=( + "Reference video input is not supported through litellm: MiniMax bills its duration, which is only " + "known after the task is created and billed. Use image or audio references instead." + ), + model=model, + llm_provider="minimax", + ) + request_data: Final[dict[str, object]] = { # mutable-ok: request body dict, JSON-serialized by the handler + "model": model, + "content": content, + } + for key in _CREATE_BODY_KEYS: + if video_create_optional_request_params.get(key) is not None: + request_data[key] = video_create_optional_request_params[key] + + request_data.setdefault("resolution", MINIMAX_VIDEO_DEFAULT_RESOLUTION) + request_data.setdefault("duration", MINIMAX_VIDEO_DEFAULT_DURATION_SECONDS) + if "ratio" not in request_data and _is_text_only_content(content): + request_data["ratio"] = MINIMAX_TEXT_TO_VIDEO_DEFAULT_RATIO + + return request_data, (), f"{api_base}/v2/video_generation" + + @staticmethod + def _content_param(video_create_optional_request_params: Mapping[str, object], prompt: str) -> object: + """ + ``prompt`` is what guardrails scanned, so it is always the one text item MiniMax takes; a caller's + ``content`` array only contributes media, never text that would bypass that check. + """ + explicit_content: Final = video_create_optional_request_params.get("content") + if explicit_content is not None: + try: + media: Final = _CALLER_MEDIA_ADAPTER.validate_python(explicit_content) + except ValidationError as e: + raise UnsupportedParamsError( + message=( + "content must be a list of MiniMax image_url, video_url or audio_url items; pass the text of " + f"the request as prompt. {e.error_count()} invalid item field(s)." + ), + llm_provider="minimax", + ) from e + return [ # mutable-ok: JSON request-body content + {"type": "text", "text": prompt}, + *(item.model_dump(exclude_none=True) for item in media), + ] + + content_items: Final[list[dict[str, object]]] = [ # mutable-ok: JSON request-body content items + {"type": "text", "text": prompt} # mutable-ok: JSON request-body content item + ] + input_reference: Final = video_create_optional_request_params.get("input_reference") + if input_reference is not None: + content_items.append(_first_frame_content_item(input_reference)) + return content_items + + def transform_video_create_response( + self, + model: str, + raw_response: httpx.Response, + logging_obj: "Logging", + custom_llm_provider: str | None = None, + request_data: dict[str, object] | None = None, # mutable-ok: BaseVideoConfig contract + ) -> VideoObject: + task_id: Final = _parse_task_id_response(raw_response) + request_mapping: Final[Mapping[str, object]] = request_data if request_data is not None else _EMPTY_PARAMS + duration: Final = request_mapping.get("duration") + resolution: Final = request_mapping.get("resolution") + input_image_count: Final = sum( + 1 for item in _content_items(request_mapping.get("content")) if item.type == "image_url" + ) + + video_obj: Final = VideoObject( + id=task_id, + object="video", + status="queued", + model=model, + seconds=str(duration) if duration is not None else None, + ) + if custom_llm_provider and video_obj.id: + video_obj.id = encode_video_id_with_provider(video_obj.id, custom_llm_provider, model) + + usage: Final[dict[str, object]] = {} # mutable-ok: VideoObject.usage is a mutable dict field + if isinstance(duration, (int, float)): + usage["duration_seconds"] = float(duration) + if isinstance(resolution, str): + usage["video_resolution"] = resolution.strip().lower() + if input_image_count: + usage["input_image_count"] = input_image_count + create_cost: Final = ( + _create_cost_usd(model, float(duration), resolution.strip().lower(), input_image_count) + if isinstance(duration, (int, float)) and isinstance(resolution, str) + else None + ) + if create_cost is not None: + usage["provider_reported_cost_usd"] = create_cost + video_obj.usage = usage + + return video_obj + + def transform_video_status_retrieve_request( + self, + video_id: str, + api_base: str, + litellm_params: GenericLiteLLMParams, + headers: dict[str, str], # mutable-ok: BaseVideoConfig contract + ) -> tuple[str, dict[str, object]]: # mutable-ok: BaseVideoConfig contract + original_task_id: Final = extract_original_video_id(video_id) + encoded_task_id: Final = encode_url_path_segment(original_task_id, field_name="video_id") + return f"{api_base}/v2/query/video_generation/{encoded_task_id}", _EMPTY_PARAMS + + def transform_video_status_retrieve_response( + self, + raw_response: httpx.Response, + logging_obj: "Logging", + custom_llm_provider: str | None = None, + client: "HTTPHandler | None" = None, + ) -> VideoObject: + task: Final = _parse_task_response(raw_response) + video_obj: Final = _video_object_from_task(task) + if custom_llm_provider and video_obj.id: + video_obj.id = encode_video_id_with_provider(video_obj.id, custom_llm_provider, task.model) + return video_obj + + def transform_video_content_request( + self, + video_id: str, + api_base: str, + litellm_params: GenericLiteLLMParams, + headers: dict[str, str], # mutable-ok: BaseVideoConfig contract + variant: str | None = None, + ) -> tuple[str, dict[str, object]]: # mutable-ok: BaseVideoConfig contract + original_task_id: Final = extract_original_video_id(video_id) + encoded_task_id: Final = encode_url_path_segment(original_task_id, field_name="video_id") + return f"{api_base}/v2/query/video_generation/{encoded_task_id}", _EMPTY_PARAMS + + def transform_video_content_response( + self, + raw_response: httpx.Response, + logging_obj: "Logging", + ) -> bytes: + task: Final = _parse_task_response(raw_response) + video_url: Final = _video_url_from_task(task) + + httpx_client: Final = _get_httpx_client() + video_response: Final = httpx_client.get(video_url) # pyright: ignore[reportUnknownMemberType] # HTTPHandler.get stub leaves params/headers untyped + video_response.raise_for_status() + + return video_response.content + + async def async_transform_video_content_response( + self, + raw_response: httpx.Response, + logging_obj: "Logging", + ) -> bytes: + task: Final = _parse_task_response(raw_response) + video_url: Final = _video_url_from_task(task) + + async_httpx_client: Final = get_async_httpx_client( + llm_provider=litellm.LlmProviders.MINIMAX, + ) + video_response: Final = await async_httpx_client.get(video_url) # pyright: ignore[reportUnknownMemberType] # HTTPHandler.get stub leaves params/headers untyped + video_response.raise_for_status() + + return video_response.content + + def transform_video_remix_request( + self, + video_id: str, + prompt: str, + api_base: str, + litellm_params: GenericLiteLLMParams, + headers: dict[str, str], # mutable-ok: BaseVideoConfig contract + extra_body: dict[str, object] | None = None, # mutable-ok: BaseVideoConfig contract + ) -> tuple[str, dict[str, object]]: # mutable-ok: BaseVideoConfig contract + raise NotImplementedError( + "Video remix is not supported by MiniMax. Its regeneration endpoint only upscales a finished 768P " + "MiniMax-H3 task to 2K and ignores the prompt; send a new video_generation() request with the edited " + "prompt instead." + ) + + def transform_video_remix_response( + self, + raw_response: httpx.Response, + logging_obj: "Logging", + custom_llm_provider: str | None = None, + ) -> VideoObject: + raise NotImplementedError("Video remix is not supported by MiniMax.") + + def transform_video_list_request( + self, + api_base: str, + litellm_params: GenericLiteLLMParams, + headers: dict[str, str], # mutable-ok: BaseVideoConfig contract + after: str | None = None, + limit: int | None = None, + order: str | None = None, + extra_query: dict[str, object] | None = None, # mutable-ok: BaseVideoConfig contract + ) -> tuple[str, dict[str, object]]: # mutable-ok: BaseVideoConfig contract + if after is not None: + raise UnsupportedParamsError( + message=( + "MiniMax video list does not support cursor pagination via 'after'. " + "Pass extra_query={'page_num': N} to request a later page." + ), + llm_provider="minimax", + ) + if order is not None and order != "desc": + raise UnsupportedParamsError( + message="MiniMax video list only returns newest first; order must be 'desc' or omitted.", + llm_provider="minimax", + ) + params: Final[dict[str, object]] = {} # mutable-ok: query params dict consumed by the http handler + if limit is not None: + params["page_size"] = str(limit) + if extra_query: + params.update(extra_query) + return f"{api_base}/v2/query/video_generation", params + + def transform_video_list_response( # pyright: ignore[reportIncompatibleMethodOverride] # base declares dict[str, str] but the payload is a heterogeneous list body + self, + raw_response: httpx.Response, + logging_obj: "Logging", + custom_llm_provider: str | None = None, + ) -> dict[str, object]: # mutable-ok: OpenAI list body served as JSON by the proxy + list_payload: Final = _LIST_RESPONSE_ADAPTER.validate_python(raw_response.json()) + total: Final = list_payload.total + + data: Final[list[dict[str, object]]] = [] # mutable-ok: OpenAI list body served as JSON by the proxy + for task in list_payload.items or (): + video_obj = _video_object_from_task(task) + if custom_llm_provider and video_obj.id: + video_obj.id = encode_video_id_with_provider(video_obj.id, custom_llm_provider, video_obj.model) + data.append(video_obj.model_dump()) + + list_response: Final[dict[str, object]] = { # mutable-ok: OpenAI list body served as JSON by the proxy + "object": "list", + "data": data, + "total": total if isinstance(total, int) and not isinstance(total, bool) else len(data), + } + if data: + list_response["first_id"] = data[0]["id"] + list_response["last_id"] = data[-1]["id"] + return list_response + + def transform_video_delete_request( + self, + video_id: str, + api_base: str, + litellm_params: GenericLiteLLMParams, + headers: dict[str, str], # mutable-ok: BaseVideoConfig contract + ) -> tuple[str, dict[str, object]]: # mutable-ok: BaseVideoConfig contract + original_task_id: Final = extract_original_video_id(video_id) + encoded_task_id: Final = encode_url_path_segment(original_task_id, field_name="video_id") + return f"{api_base}/v2/video_generation/{encoded_task_id}", _EMPTY_PARAMS + + def transform_video_delete_response( + self, + raw_response: httpx.Response, + logging_obj: "Logging", + ) -> VideoObject: + task_id, status = _parse_delete_response(raw_response) + return VideoObject( + id=task_id, + object="video", + status=status, + ) + + def get_error_class( + self, + error_message: str, + status_code: int, + headers: dict[str, str] | httpx.Headers, # mutable-ok: BaseVideoConfig contract + ) -> BaseLLMException: + return MinimaxVideoError( + status_code=status_code, + message=error_message, + headers=headers, + ) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index fc48c17b506..424622b87ff 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -37986,6 +37986,60 @@ "max_input_tokens": 1000000, "max_output_tokens": 128000 }, + "minimax/MiniMax-H3": { + "litellm_provider": "minimax", + "mode": "video_generation", + "source": "https://platform.minimax.io/docs/guides/pricing-paygo", + "output_cost_per_second": 0.08, + "output_cost_per_second_768p": 0.08, + "output_cost_per_second_2k": 0.13, + "input_cost_per_image": 0.04, + "provider_specific_entry": { + "minimax_free_input_images": 5 + }, + "supported_modalities": [ + "text", + "image", + "video", + "audio" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "metadata": { + "comment": "V2 async task API. Pay-as-you-go list price $0.08/s at 768P and $0.13/s at 2K, plus $0.04 per input image beyond the first 5. Audio input is free. Reference video input is rejected, since MiniMax bills its duration only after the create call." + } + }, + "minimax/MiniMax-H3-Max": { + "litellm_provider": "minimax", + "mode": "video_generation", + "source": "https://platform.minimax.io/docs/guides/pricing-paygo", + "output_cost_per_second": 0.08, + "output_cost_per_second_480p": 0.05, + "output_cost_per_second_768p": 0.08, + "input_cost_per_image": 0.074, + "provider_specific_entry": { + "minimax_free_input_images": 2 + }, + "supported_modalities": [ + "text", + "image", + "video", + "audio" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "metadata": { + "comment": "V2 async task API, fast tier. Pay-as-you-go list price $0.05/s at 480P and $0.08/s at 768P; no 2K. Plus $0.074 per input image beyond the first 2. Audio input is free. Reference video input is rejected, since MiniMax bills its duration only after the create call." + } + }, "mistral.devstral-2-123b": { "input_cost_per_token": 4e-07, "litellm_provider": "bedrock_converse", diff --git a/litellm/utils.py b/litellm/utils.py index 13a46840431..90f72787039 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -9638,6 +9638,10 @@ class ProviderConfigManager: from litellm.llms.hosted_vllm.videos import get_hosted_vllm_video_config return get_hosted_vllm_video_config(model) + elif LlmProviders.MINIMAX == provider: + from litellm.llms.minimax.videos.transformation import MinimaxVideoConfig + + return MinimaxVideoConfig() elif LlmProviders.EDENAI == provider: return litellm.EdenAIVideoConfig() return None diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index fc48c17b506..424622b87ff 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -37986,6 +37986,60 @@ "max_input_tokens": 1000000, "max_output_tokens": 128000 }, + "minimax/MiniMax-H3": { + "litellm_provider": "minimax", + "mode": "video_generation", + "source": "https://platform.minimax.io/docs/guides/pricing-paygo", + "output_cost_per_second": 0.08, + "output_cost_per_second_768p": 0.08, + "output_cost_per_second_2k": 0.13, + "input_cost_per_image": 0.04, + "provider_specific_entry": { + "minimax_free_input_images": 5 + }, + "supported_modalities": [ + "text", + "image", + "video", + "audio" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "metadata": { + "comment": "V2 async task API. Pay-as-you-go list price $0.08/s at 768P and $0.13/s at 2K, plus $0.04 per input image beyond the first 5. Audio input is free. Reference video input is rejected, since MiniMax bills its duration only after the create call." + } + }, + "minimax/MiniMax-H3-Max": { + "litellm_provider": "minimax", + "mode": "video_generation", + "source": "https://platform.minimax.io/docs/guides/pricing-paygo", + "output_cost_per_second": 0.08, + "output_cost_per_second_480p": 0.05, + "output_cost_per_second_768p": 0.08, + "input_cost_per_image": 0.074, + "provider_specific_entry": { + "minimax_free_input_images": 2 + }, + "supported_modalities": [ + "text", + "image", + "video", + "audio" + ], + "supported_output_modalities": [ + "video" + ], + "supported_endpoints": [ + "/v1/videos" + ], + "metadata": { + "comment": "V2 async task API, fast tier. Pay-as-you-go list price $0.05/s at 480P and $0.08/s at 768P; no 2K. Plus $0.074 per input image beyond the first 2. Audio input is free. Reference video input is rejected, since MiniMax bills its duration only after the create call." + } + }, "mistral.devstral-2-123b": { "input_cost_per_token": 4e-07, "litellm_provider": "bedrock_converse", diff --git a/tests/unit/llms/minimax/videos/__init__.py b/tests/unit/llms/minimax/videos/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/tests/unit/llms/minimax/videos/test_minimax_video_transformation.py b/tests/unit/llms/minimax/videos/test_minimax_video_transformation.py new file mode 100644 index 00000000000..51e5e0d0aa7 --- /dev/null +++ b/tests/unit/llms/minimax/videos/test_minimax_video_transformation.py @@ -0,0 +1,697 @@ +""" +Tests for MiniMax (Hailuo-03) video generation transformation. +""" + +import base64 +import io +import json +from typing import Final +from unittest.mock import Mock + +import httpx +import pytest + +from litellm.exceptions import UnsupportedParamsError +from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler +from litellm.llms.minimax.videos.transformation import ( + _MiniMaxTask, + _TaskContent, + _TaskError, + _video_url_from_task, + MinimaxVideoConfig, +) +from litellm.types.router import GenericLiteLLMParams +from litellm.types.videos.utils import ( + decode_video_id_with_provider, + encode_video_id_with_provider, +) +from litellm.videos.main import avideo_generation + +PNG_BYTES = base64.b64decode( + "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg==" +) + +API_BASE = "https://api.minimax.io" + + +def _mock_response(payload: dict) -> Mock: + mock_response = Mock(spec=httpx.Response) + mock_response.json.return_value = payload + return mock_response + + +def _query_response(task: dict) -> Mock: + return _mock_response({"task": task}) + + +class TestMinimaxVideoCreateRequest: + def test_text_to_video_defaults(self): + """A prompt-only request must build the content array and apply + MiniMax's required resolution/duration/ratio when the caller omits them.""" + data, files, url = MinimaxVideoConfig().transform_video_create_request( + model="MiniMax-H3", + prompt="A cinematic shot of a lighthouse at dusk", + api_base=API_BASE, + video_create_optional_request_params={}, + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + assert url == f"{API_BASE}/v2/video_generation" + assert files == () + assert data["model"] == "MiniMax-H3" + assert data["content"] == [{"type": "text", "text": "A cinematic shot of a lighthouse at dusk"}] + assert data["resolution"] == "768P" + assert data["duration"] == 5 + assert data["ratio"] == "16:9" + + def test_explicit_params_beat_defaults(self): + data, _, _ = MinimaxVideoConfig().transform_video_create_request( + model="MiniMax-H3-Max", + prompt="prompt", + api_base=API_BASE, + video_create_optional_request_params={"resolution": "480P", "duration": 9, "ratio": "9:16"}, + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + assert data["model"] == "MiniMax-H3-Max" + assert data["resolution"] == "480P" + assert data["duration"] == 9 + assert data["ratio"] == "9:16" + + def test_map_openai_params_converts_seconds_and_size(self): + """OpenAI ``seconds``/``size`` must become MiniMax ``duration``/``ratio`` + (1280x720 reduces to 16:9), not be forwarded verbatim.""" + mapped = MinimaxVideoConfig().map_openai_params( + video_create_optional_params={"seconds": "5", "size": "1280x720", "resolution": "2K"}, + model="MiniMax-H3", + drop_params=False, + ) + + assert mapped["duration"] == 5 + assert mapped["ratio"] == "16:9" + assert mapped["resolution"] == "2K" + assert "seconds" not in mapped + assert "size" not in mapped + + def test_map_openai_params_drops_fields_minimax_rejects(self): + mapped = MinimaxVideoConfig().map_openai_params( + video_create_optional_params={"user": "u1", "characters": [{"id": "c"}], "prompt": "p"}, + model="MiniMax-H3", + drop_params=False, + ) + + assert mapped == {} + + def test_map_openai_params_explicit_ratio_wins_over_size(self): + mapped = MinimaxVideoConfig().map_openai_params( + video_create_optional_params={"size": "1280x720", "ratio": "4:3"}, + model="MiniMax-H3", + drop_params=False, + ) + + assert mapped["ratio"] == "4:3" + + def test_map_openai_params_merges_parameters_block(self): + mapped = MinimaxVideoConfig().map_openai_params( + video_create_optional_params={"parameters": {"callback_url": "https://cb.example/hook", "ratio": "1:1"}}, + model="MiniMax-H3", + drop_params=False, + ) + + assert mapped == {"callback_url": "https://cb.example/hook", "ratio": "1:1"} + + def test_image_reference_file_becomes_first_frame_data_uri(self): + """A file input_reference must arrive as a first_frame content item + carrying a base64 data URI, and text-only ratio defaults must not apply.""" + data, _, _ = MinimaxVideoConfig().transform_video_create_request( + model="MiniMax-H3", + prompt="Pull focus to the people in the background", + api_base=API_BASE, + video_create_optional_request_params={"input_reference": io.BytesIO(PNG_BYTES)}, + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + image_item = data["content"][1] + assert image_item["type"] == "image_url" + assert image_item["role"] == "first_frame" + assert image_item["image_url"]["url"].startswith("data:image/png;base64,") + encoded = image_item["image_url"]["url"].split(",", 1)[1] + assert base64.b64decode(encoded) == PNG_BYTES + assert "ratio" not in data + + def test_image_reference_url_passthrough(self): + data, _, _ = MinimaxVideoConfig().transform_video_create_request( + model="MiniMax-H3", + prompt="Add more steam to the ramen bowl", + api_base=API_BASE, + video_create_optional_request_params={"input_reference": "https://cdn.example.com/frame.png"}, + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + assert data["content"][1] == { + "type": "image_url", + "image_url": {"url": "https://cdn.example.com/frame.png"}, + "role": "first_frame", + } + + def test_explicit_media_content_follows_the_prompt(self): + """Multimodal-reference (r2va) callers supply the media items; they pass + through after the prompt and suppress the text-only ratio default.""" + media = [ + {"type": "image_url", "image_url": {"url": "https://cdn.example.com/ref.png"}, "role": "reference_image"}, + {"type": "audio_url", "audio_url": {"url": "https://cdn.example.com/ref.mp3"}, "role": "reference_audio"}, + ] + data, _, _ = MinimaxVideoConfig().transform_video_create_request( + model="MiniMax-H3", + prompt="Character speaking", + api_base=API_BASE, + video_create_optional_request_params={"content": media}, + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + assert data["content"] == [{"type": "text", "text": "Character speaking"}, *media] + assert "ratio" not in data + + @pytest.mark.parametrize( + "content", + [ + [{"type": "text", "text": "unscanned text"}, {"type": "image_url"}], + [{"type": "image_url", "image_url": {"url": "https://x/a.png"}, "text": "unscanned text"}], + [{"type": "image_url", "image_url": {"url": "https://x/a.png", "text": "unscanned text"}}], + [{"type": "text", "text": "unscanned text"}, "not an item"], + "unscanned text", + ], + ) + def test_malformed_content_cannot_smuggle_unscanned_text(self, content): + """A text item hidden behind a malformed sibling, or text on a media + item, must be rejected rather than let through by a lenient parse.""" + with pytest.raises(UnsupportedParamsError) as raised: + MinimaxVideoConfig().transform_video_create_request( + model="MiniMax-H3", + prompt="a calm lake", + api_base=API_BASE, + video_create_optional_request_params={"content": content}, + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + assert raised.value.status_code == 400 + + def test_text_inside_content_is_rejected_so_guardrails_cannot_be_bypassed(self): + """Guardrails scan prompt; a text item smuggled into content would reach + MiniMax unscanned while a harmless prompt passed the check.""" + with pytest.raises(UnsupportedParamsError, match="as prompt") as raised: + MinimaxVideoConfig().transform_video_create_request( + model="MiniMax-H3", + prompt="a calm lake", + api_base=API_BASE, + video_create_optional_request_params={"content": [{"type": "text", "text": "unscanned text"}]}, + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + assert raised.value.status_code == 400 + + @pytest.mark.parametrize( + "configured_api_base", + ( + "https://api.minimax.cn/v1/", + "https://api.minimax.cn/v1", + "https://api.minimax.cn/anthropic", + "https://api.minimax.cn/anthropic/", + "https://api.minimax.cn", + "https://api.minimax.cn/", + "https://api.minimax.cn/v2", + ), + ) + def test_any_published_surface_base_reaches_video(self, configured_api_base): + """A MiniMax key works across that host's API surfaces, so an existing + chat credential must reach video whichever base it was configured with: + MiniMax publishes .../v1 (OpenAI-compatible) and .../anthropic + (Anthropic Messages) alongside the /v2 video API, and users also + configure the bare host.""" + config = MinimaxVideoConfig() + api_base = config.get_complete_url(model="MiniMax-H3", api_base=configured_api_base, litellm_params={}) + + _, _, url = config.transform_video_create_request( + model="MiniMax-H3", + prompt="p", + api_base=api_base, + video_create_optional_request_params={}, + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + assert url == "https://api.minimax.cn/v2/video_generation" + + @pytest.mark.parametrize( + ("configured_api_base", "expected"), + ( + ("https://anthropic.example.com/v1", "https://anthropic.example.com"), + ("https://api.minimax.cn/v1/proxy", "https://api.minimax.cn/v1/proxy"), + ("https://gateway.internal/anthropic/shim", "https://gateway.internal/anthropic/shim"), + ), + ) + def test_surface_suffix_is_only_stripped_from_the_end(self, configured_api_base, expected): + """Only a trailing surface segment marks the protocol. One appearing + anywhere else is part of the address: cutting it out of the middle sends + the request to a different path than the operator configured.""" + config = MinimaxVideoConfig() + + assert config.get_complete_url(model="MiniMax-H3", api_base=configured_api_base, litellm_params={}) == expected + + def test_get_complete_url_defaults_to_international_host(self): + assert MinimaxVideoConfig().get_complete_url(model="MiniMax-H3", api_base=None, litellm_params={}) == ( + "https://api.minimax.io" + ) + + +class TestMinimaxVideoCreateResponse: + def test_task_id_is_encoded_with_provider_and_model(self): + """The create response only carries task_id; litellm must wrap it so + later status/content/remix calls can route back to minimax.""" + video_obj = MinimaxVideoConfig().transform_video_create_response( + model="MiniMax-H3", + raw_response=_mock_response({"task_id": "424010985738629"}), + logging_obj=None, + custom_llm_provider="minimax", + request_data={"model": "MiniMax-H3", "content": [], "resolution": "2K", "duration": 5, "ratio": "16:9"}, + ) + + assert video_obj.status == "queued" + assert video_obj.model == "MiniMax-H3" + assert video_obj.seconds == "5" + decoded = decode_video_id_with_provider(video_obj.id) + assert decoded["custom_llm_provider"] == "minimax" + assert decoded["model_id"] == "MiniMax-H3" + assert decoded["video_id"] == "424010985738629" + + def test_usage_carries_cost_inputs(self): + video_obj = MinimaxVideoConfig().transform_video_create_response( + model="MiniMax-H3", + raw_response=_mock_response({"task_id": "t1"}), + logging_obj=None, + custom_llm_provider="minimax", + request_data={"resolution": "768P", "duration": 4}, + ) + + assert video_obj.usage["duration_seconds"] == 4.0 + assert video_obj.usage["video_resolution"] == "768p" + assert "input_image_count" not in video_obj.usage + + +class TestMinimaxVideoStatus: + def test_succeeded_task_mapping(self): + video_obj = MinimaxVideoConfig().transform_video_status_retrieve_response( + raw_response=_query_response( + { + "id": "424010985738629", + "model": "MiniMax-H3", + "status": "succeeded", + "created_at": 1785125529, + "updated_at": 1785125946, + "content": {"url": "https://cdn.example.com/output.mp4"}, + "resolution": "2K", + "duration": 5, + "usage": {"total_seconds": 5, "input_seconds": 0, "output_seconds": 5, "input_image_count": 1}, + "ratio": "16:9", + "task_type": "generation", + "modality": "video", + } + ), + logging_obj=None, + custom_llm_provider="minimax", + ) + + assert video_obj.status == "completed" + assert video_obj.created_at == 1785125529 + assert video_obj.completed_at == 1785125946 + assert video_obj.seconds == "5" + assert video_obj.model == "MiniMax-H3" + assert video_obj.usage["output_seconds"] == 5 + decoded = decode_video_id_with_provider(video_obj.id) + assert decoded["custom_llm_provider"] == "minimax" + assert decoded["model_id"] == "MiniMax-H3" + + def test_running_task_maps_to_in_progress_without_completion(self): + video_obj = MinimaxVideoConfig().transform_video_status_retrieve_response( + raw_response=_query_response({"id": "t1", "model": "MiniMax-H3", "status": "running", "created_at": 1}), + logging_obj=None, + custom_llm_provider="minimax", + ) + + assert video_obj.status == "in_progress" + assert video_obj.completed_at is None + assert video_obj.usage is None + + def test_failed_task_maps_error(self): + video_obj = MinimaxVideoConfig().transform_video_status_retrieve_response( + raw_response=_query_response( + { + "id": "t1", + "status": "failed", + "error": {"code": "1026", "message": "video description contains sensitive content"}, + "created_at": 1, + } + ), + logging_obj=None, + custom_llm_provider="minimax", + ) + + assert video_obj.status == "failed" + assert video_obj.error == {"code": "1026", "message": "video description contains sensitive content"} + + def test_request_decodes_task_id_from_wrapped_video_id(self): + encoded_video_id = encode_video_id_with_provider("424010985738629", "minimax", "MiniMax-H3") + url, data = MinimaxVideoConfig().transform_video_status_retrieve_request( + video_id=encoded_video_id, + api_base=API_BASE, + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + assert url == f"{API_BASE}/v2/query/video_generation/424010985738629" + assert data == {} + + def test_content_request_hits_query_endpoint(self): + encoded_video_id = encode_video_id_with_provider("424010985738629", "minimax", "MiniMax-H3") + url, data = MinimaxVideoConfig().transform_video_content_request( + video_id=encoded_video_id, + api_base=API_BASE, + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + assert url == f"{API_BASE}/v2/query/video_generation/424010985738629" + + def test_video_url_from_task_pending_and_failed_raise(self): + with pytest.raises(ValueError, match="still processing"): + _video_url_from_task(_MiniMaxTask(status="running")) + with pytest.raises(ValueError, match="sensitive content"): + _video_url_from_task(_MiniMaxTask(status="failed", error=_TaskError(message="sensitive content"))) + + +class TestMinimaxVideoList: + def test_request_maps_limit_and_extra_query(self): + url, params = MinimaxVideoConfig().transform_video_list_request( + api_base=API_BASE, + litellm_params=GenericLiteLLMParams(), + headers={}, + limit=4, + extra_query={"filter.status": "succeeded", "page_num": 2}, + ) + + assert url == f"{API_BASE}/v2/query/video_generation" + assert params == {"page_size": "4", "filter.status": "succeeded", "page_num": 2} + + def test_after_cursor_is_rejected_instead_of_repeating_the_first_page(self): + with pytest.raises(UnsupportedParamsError, match="page_num") as raised: + MinimaxVideoConfig().transform_video_list_request( + api_base=API_BASE, + litellm_params=GenericLiteLLMParams(), + headers={}, + after="video_abc", + ) + + assert raised.value.status_code == 400 + + def test_ascending_order_is_rejected(self): + with pytest.raises(UnsupportedParamsError, match="newest first") as raised: + MinimaxVideoConfig().transform_video_list_request( + api_base=API_BASE, + litellm_params=GenericLiteLLMParams(), + headers={}, + order="asc", + ) + + assert raised.value.status_code == 400 + + def test_descending_order_is_what_minimax_already_returns(self): + _, params = MinimaxVideoConfig().transform_video_list_request( + api_base=API_BASE, + litellm_params=GenericLiteLLMParams(), + headers={}, + order="desc", + ) + + assert params == {} + + def test_response_adopts_openai_list_shape_with_encoded_ids(self): + response = MinimaxVideoConfig().transform_video_list_response( + raw_response=_mock_response( + { + "items": [ + {"id": "424635601932571", "model": "MiniMax-H3", "status": "succeeded", "duration": 5}, + {"id": "424635601932588", "model": "MiniMax-H3", "status": "running"}, + ], + "total": 476, + } + ), + logging_obj=None, + custom_llm_provider="minimax", + ) + + assert response["object"] == "list" + assert response["total"] == 476 + assert [item["status"] for item in response["data"]] == ["completed", "in_progress"] + first_decoded = decode_video_id_with_provider(response["first_id"]) + last_decoded = decode_video_id_with_provider(response["last_id"]) + assert first_decoded["video_id"] == "424635601932571" + assert last_decoded["video_id"] == "424635601932588" + assert first_decoded["custom_llm_provider"] == "minimax" + + +class TestMinimaxVideoRemix: + def test_remix_is_rejected_instead_of_silently_dropping_the_prompt(self): + """MiniMax regeneration only upscales to 2K and ignores any prompt, so + mapping remix onto it would return a video that ignores the edit.""" + encoded_video_id = encode_video_id_with_provider("424010985738629", "minimax", "MiniMax-H3") + + with pytest.raises(NotImplementedError, match="remix is not supported by MiniMax"): + MinimaxVideoConfig().transform_video_remix_request( + video_id=encoded_video_id, + prompt="a different ending", + api_base=API_BASE, + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + +class TestMinimaxVideoDelete: + def test_delete_request_and_cancelled_response(self): + encoded_video_id = encode_video_id_with_provider("424010985738629", "minimax", "MiniMax-H3") + url, data = MinimaxVideoConfig().transform_video_delete_request( + video_id=encoded_video_id, + api_base=API_BASE, + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + assert url == f"{API_BASE}/v2/video_generation/424010985738629" + assert data == {} + + video_obj = MinimaxVideoConfig().transform_video_delete_response( + raw_response=_mock_response({"task_id": "424010985738629", "action": "cancelled", "status": "cancelled"}), + logging_obj=None, + ) + assert video_obj.status == "cancelled" + assert video_obj.id == "424010985738629" + + +class TestMinimaxVideoEnvironment: + def test_explicit_api_key_wins_over_litellm_params(self): + headers = MinimaxVideoConfig().validate_environment( + headers={}, + model="MiniMax-H3", + api_key="explicit-key", + litellm_params=GenericLiteLLMParams(api_key="deployment-key"), + ) + + assert headers["Authorization"] == "Bearer explicit-key" + assert headers["Content-Type"] == "application/json" + + def test_litellm_params_key_used_when_no_explicit_key(self): + headers = MinimaxVideoConfig().validate_environment( + headers={}, + model="MiniMax-H3", + litellm_params=GenericLiteLLMParams(api_key="deployment-key"), + ) + + assert headers["Authorization"] == "Bearer deployment-key" + + def test_missing_api_key_raises(self, monkeypatch): + monkeypatch.delenv("MINIMAX_API_KEY", raising=False) + monkeypatch.setattr("litellm.api_key", None) + + with pytest.raises(ValueError, match="MINIMAX_API_KEY"): + MinimaxVideoConfig().validate_environment( + headers={}, + model="MiniMax-H3", + litellm_params=GenericLiteLLMParams(), + ) + + def test_env_var_api_key_used(self, monkeypatch): + monkeypatch.setenv("MINIMAX_API_KEY", "env-key") + monkeypatch.setattr("litellm.api_key", None) + + headers = MinimaxVideoConfig().validate_environment( + headers={}, + model="MiniMax-H3", + litellm_params=GenericLiteLLMParams(), + ) + + assert headers["Authorization"] == "Bearer env-key" + + +class TestMinimaxVideoEndToEndRequest: + """ + Drive the real avideo_generation() entrypoint rather than the transform. + + map_openai_params runs first and only its output reaches + transform_video_create_request, so a param the mapper drops never makes + it into the body even though the transform alone handles it. + """ + + @staticmethod + async def _wire_body(**kwargs) -> dict: + sent: Final[list[httpx.Request]] = [] + + def respond(request: httpx.Request) -> httpx.Response: + sent.append(request) + return httpx.Response(200, json={"task_id": "t1"}) + + async with httpx.AsyncClient(transport=httpx.MockTransport(respond)) as http_client: + handler: Final = AsyncHTTPHandler() + await handler.close() + handler.client = http_client + await avideo_generation(api_key="sk-test", api_base=API_BASE, client=handler, **kwargs) + + (request,) = sent + return json.loads(request.content) + + @pytest.mark.asyncio + async def test_input_reference_survives_param_mapping_into_the_body(self): + """Regression: map_openai_params dropped input_reference, so an + image-to-video call silently degraded to text-to-video.""" + body = await self._wire_body( + model="minimax/MiniMax-H3", + prompt="make it move", + input_reference="https://cdn.example/first.png", + ) + + assert body["content"] == [ + {"type": "text", "text": "make it move"}, + {"type": "image_url", "image_url": {"url": "https://cdn.example/first.png"}, "role": "first_frame"}, + ] + assert "input_reference" not in body + assert "ratio" not in body + + @pytest.mark.asyncio + async def test_file_input_reference_reaches_the_body_as_a_data_uri(self): + body = await self._wire_body( + model="minimax/MiniMax-H3", + prompt="make it move", + input_reference=io.BytesIO(PNG_BYTES), + ) + + image_item = body["content"][1] + assert image_item["role"] == "first_frame" + assert image_item["image_url"]["url"] == f"data:image/png;base64,{base64.b64encode(PNG_BYTES).decode()}" + + +@pytest.mark.usefixtures("local_model_cost_map") +class TestMinimaxVideoPricing: + @pytest.mark.parametrize( + "model,resolutions", + [("MiniMax-H3", ("768p", "2k")), ("MiniMax-H3-Max", ("480p", "768p"))], + ) + def test_every_supported_tier_bills_its_own_rate(self, model, resolutions): + """A tier with no rate of its own falls back to the base rate, so a 2K + video would silently bill at the 768P price.""" + from litellm import get_model_info + from litellm.llms.openai.cost_calculation import video_generation_cost + + info = get_model_info(model=model, custom_llm_provider="minimax") + + for resolution in resolutions: + tier_rate = info[f"output_cost_per_second_{resolution}"] + assert tier_rate > 0 + cost = video_generation_cost( + model=model, duration_seconds=5.0, custom_llm_provider="minimax", video_resolution=resolution + ) + assert cost == pytest.approx(tier_rate * 5.0) + + def test_higher_resolution_never_bills_less(self): + from litellm import get_model_info + + h3 = get_model_info(model="MiniMax-H3", custom_llm_provider="minimax") + h3_max = get_model_info(model="MiniMax-H3-Max", custom_llm_provider="minimax") + + assert h3["output_cost_per_second_2k"] > h3["output_cost_per_second_768p"] + assert h3_max["output_cost_per_second_768p"] > h3_max["output_cost_per_second_480p"] + + +@pytest.mark.usefixtures("local_model_cost_map") +class TestMinimaxVideoInputBilling: + @staticmethod + def _create_cost(model: str, image_count: int) -> float: + import litellm + + content = [{"type": "text", "text": "p"}] + [ + {"type": "image_url", "image_url": {"url": f"https://x/{i}.png"}, "role": "reference_image"} + for i in range(image_count) + ] + video_obj = MinimaxVideoConfig().transform_video_create_response( + model=model, + raw_response=_mock_response({"task_id": "t1"}), + logging_obj=None, + custom_llm_provider="minimax", + request_data={"model": model, "content": content, "duration": 5, "resolution": "768P"}, + ) + return litellm.completion_cost( + completion_response=video_obj, + model=f"minimax/{model}", + call_type="create_video", + custom_llm_provider="minimax", + ) + + @pytest.mark.parametrize("model", ["MiniMax-H3", "MiniMax-H3-Max"]) + def test_images_beyond_the_free_allowance_are_billed_per_image(self, model): + """MiniMax charges for input images past a per-model free allowance; + billing only output seconds under-recorded those requests.""" + from litellm import get_model_info + + info = get_model_info(model=model, custom_llm_provider="minimax") + free_images = info["provider_specific_entry"]["minimax_free_input_images"] + image_rate = info["input_cost_per_image"] + output_only = self._create_cost(model, image_count=0) + + assert self._create_cost(model, image_count=free_images) == pytest.approx(output_only) + assert self._create_cost(model, image_count=free_images + 3) == pytest.approx(output_only + 3 * image_rate) + assert output_only == pytest.approx(info["output_cost_per_second_768p"] * 5) + + @pytest.mark.parametrize( + "content", + [ + [{"type": "video_url", "video_url": {"url": "https://x/ref.mp4"}, "role": "reference_video"}], + ], + ) + def test_reference_video_is_rejected_because_its_length_cannot_be_billed(self, content): + """MiniMax bills reference-video seconds, which are only reported after + the create call that litellm bills.""" + with pytest.raises(UnsupportedParamsError, match="Reference video") as raised: + MinimaxVideoConfig().transform_video_create_request( + model="MiniMax-H3", + prompt="p", + api_base=API_BASE, + video_create_optional_request_params={"content": content}, + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + assert raised.value.status_code == 400