From 5f54f87d9887c13b75c42ffca5142ef2b7bcf2f1 Mon Sep 17 00:00:00 2001 From: kerry Date: Sat, 19 Sep 2026 16:40:10 +0000 Subject: [PATCH 1/9] feat(fal_ai): add Seedance video generation via fal queue API Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/llms/fal_ai/videos/__init__.py | 3 + litellm/llms/fal_ai/videos/transformation.py | 512 ++++++++++++++++++ ...odel_prices_and_context_window_backup.json | 121 +++++ litellm/utils.py | 4 + model_prices_and_context_window.json | 121 +++++ .../test_fal_ai_video_transformation.py | 231 ++++++++ 6 files changed, 992 insertions(+) create mode 100644 litellm/llms/fal_ai/videos/__init__.py create mode 100644 litellm/llms/fal_ai/videos/transformation.py create mode 100644 tests/test_litellm/llms/fal_ai/videos/test_fal_ai_video_transformation.py diff --git a/litellm/llms/fal_ai/videos/__init__.py b/litellm/llms/fal_ai/videos/__init__.py new file mode 100644 index 00000000000..c7e8f76c75b --- /dev/null +++ b/litellm/llms/fal_ai/videos/__init__.py @@ -0,0 +1,3 @@ +from litellm.llms.fal_ai.videos.transformation import FalAIVideoConfig + +__all__ = ("FalAIVideoConfig",) diff --git a/litellm/llms/fal_ai/videos/transformation.py b/litellm/llms/fal_ai/videos/transformation.py new file mode 100644 index 00000000000..f8ebf828d68 --- /dev/null +++ b/litellm/llms/fal_ai/videos/transformation.py @@ -0,0 +1,512 @@ +import math +import time +from collections.abc import Mapping +from types import MappingProxyType +from typing import Final + +import httpx +from httpx._types import FileContent, RequestFiles +from pydantic import TypeAdapter + +from litellm.litellm_core_utils.url_utils import encode_url_path_segment +from litellm.llms.base_llm.chat.transformation import BaseLLMException +from litellm.llms.base_llm.videos.transformation import BaseVideoConfig +from litellm.llms.custom_httpx.http_handler import ( + AsyncHTTPHandler, + HTTPHandler, + _get_httpx_client, # pyright: ignore[reportPrivateUsage, reportUnknownVariableType] # shared HTTP factory is private + get_async_httpx_client, # pyright: ignore[reportUnknownVariableType] # shared HTTP factory lacks typed params +) +from litellm.secret_managers.main import get_secret_str +from litellm.types.router import GenericLiteLLMParams +from litellm.types.utils import LlmProviders +from litellm.types.videos.main import ( + CharacterObject, + VideoCreateOptionalRequestParams, + VideoObject, +) +from litellm.types.videos.utils import ( + decode_video_id_with_provider, + encode_video_id_with_provider, +) + + +class FalAIVideoError(BaseLLMException): + pass + + +_ALLOWED_ASPECT_RATIOS: Final[frozenset[str]] = frozenset({"auto", "16:9", "9:16", "1:1", "4:3", "3:4", "21:9"}) +_ALLOWED_RESOLUTIONS: Final[frozenset[str]] = frozenset({"480p", "720p", "1080p", "4k"}) +_RESOLUTION_TIERS: Final[tuple[tuple[int, str], ...]] = ( + (480, "480p"), + (720, "720p"), + (1080, "1080p"), +) +_FAL_AI_PROVIDER: Final[str] = LlmProviders.FAL_AI.value + + +def _queue_request_base_path(model: str) -> str: + segments: Final[tuple[str, ...]] = tuple(model.split("/")) + segment_count: Final[int] = 3 if segments and segments[0] in frozenset(("workflows", "comfy")) else 2 + return "/".join(segments[:segment_count]) + + +def _duration_value(value: object) -> str | None: + if isinstance(value, str) and value == "auto": + return value + if isinstance(value, bool) or not isinstance(value, (int, float, str)): + return None + try: + return str(int(float(value))) + except (TypeError, ValueError): + return None + + +def _resolution_for_height(height: int) -> str: + return next((resolution for threshold, resolution in _RESOLUTION_TIERS if height <= threshold), "4k") + + +def _size_params(size: object) -> Mapping[str, str]: + if not isinstance(size, str): + return MappingProxyType({}) + if size in _ALLOWED_RESOLUTIONS: + return MappingProxyType({"resolution": size}) + if size.count("x") != 1: + return MappingProxyType({}) + width_text, height_text = size.split("x") + if not (width_text.isdigit() and height_text.isdigit()): + return MappingProxyType({}) + width: Final[int] = int(width_text) + height: Final[int] = int(height_text) + if width <= 0 or height <= 0: + return MappingProxyType({}) + reduced_gcd: Final[int] = math.gcd(width, height) + aspect_ratio: Final[str] = f"{width // reduced_gcd}:{height // reduced_gcd}" + resolution: Final[str] = _resolution_for_height(height) + if aspect_ratio in _ALLOWED_ASPECT_RATIOS: + return MappingProxyType({"resolution": resolution, "aspect_ratio": aspect_ratio}) + return MappingProxyType({"resolution": resolution}) + + +def _numeric_duration(value: object) -> float | None: + duration: Final[str | None] = _duration_value(value) + if duration is None or duration == "auto": + return None + return float(duration) + + +def _response_data(raw_response: httpx.Response) -> Mapping[str, object]: + return TypeAdapter(Mapping[str, object]).validate_python(raw_response.json()) + + +def _response_string(response_data: Mapping[str, object], key: str, default: str = "") -> str: + value: Final[object] = response_data.get(key) + return value if isinstance(value, str) else default + + +class FalAIVideoConfig(BaseVideoConfig): + def get_supported_openai_params(self, model: str) -> list[str]: # mutable-ok: API contract requires a list + return [ # mutable-ok: API contract requires a list + "model", + "prompt", + "input_reference", + "seconds", + "size", + "user", + "extra_headers", + ] + + def map_openai_params( + self, + video_create_optional_params: VideoCreateOptionalRequestParams, + model: str, + drop_params: bool, + ) -> dict[str, object]: # mutable-ok: BaseVideoConfig requires a mutable mapping + supported_params: Final[frozenset[str]] = frozenset(self.get_supported_openai_params(model)) + input_reference: Final[object] = video_create_optional_params.get("input_reference") + input_reference_params: Final[Mapping[str, str]] = ( + MappingProxyType({}) + if "input_reference" not in video_create_optional_params + else ( + MappingProxyType({"image_url": input_reference}) + if isinstance(input_reference, str) + else self._invalid_input_reference() + ) + ) + duration_params: Final[Mapping[str, str]] = ( + MappingProxyType({}) + if "seconds" not in video_create_optional_params + else self._duration_params(video_create_optional_params["seconds"]) + ) + size_params: Final[Mapping[str, str]] = ( + self._size_params(video_create_optional_params["size"]) + if "size" in video_create_optional_params + else MappingProxyType({}) + ) + user_params: Final[Mapping[str, str]] = ( + MappingProxyType({"end_user_id": user}) + if isinstance(user := video_create_optional_params.get("user"), str) + else MappingProxyType({}) + ) + return dict( # mutable-ok: BaseVideoConfig requires a mutable mapping + MappingProxyType( + { + **input_reference_params, + **duration_params, + **size_params, + **user_params, + **{ # mutable-ok: dynamic passthrough fields require a mapping + key: value for key, value in video_create_optional_params.items() if key not in supported_params + }, + } + ) + ) # mutable-ok: BaseVideoConfig requires a mutable mapping + + @staticmethod + def _invalid_input_reference() -> Mapping[str, str]: + raise ValueError("fal.ai needs a public image URL for input_reference") + + @staticmethod + def _duration_params(seconds: object) -> Mapping[str, str]: + duration: Final[str | None] = _duration_value(seconds) + if duration is None: + raise ValueError("fal.ai seconds must be a numeric value") + return MappingProxyType({"duration": duration}) + + @staticmethod + def _size_params(size: object) -> Mapping[str, str]: + return _size_params(size) + + def validate_environment( + self, + headers: dict[str, str], # mutable-ok: BaseVideoConfig requires mutable headers + model: str, + api_key: str | None = None, + litellm_params: GenericLiteLLMParams | None = None, + ) -> dict[str, str]: # mutable-ok: BaseVideoConfig requires mutable headers + final_api_key: Final[str | None] = ( + api_key + or (litellm_params.api_key if litellm_params is not None else None) + or get_secret_str("FAL_AI_API_KEY") + or get_secret_str("FAL_KEY") + ) + if not final_api_key: + raise ValueError("fal.ai API key is required") + return dict( # mutable-ok: BaseVideoConfig requires mutable headers + MappingProxyType( + { + **headers, + "Authorization": f"Key {final_api_key}", + "Content-Type": "application/json", + } + ) + ) # mutable-ok: BaseVideoConfig requires mutable headers + + def get_complete_url( + self, + model: str, + api_base: str | None, + litellm_params: dict[str, object], # mutable-ok: BaseVideoConfig requires mutable parameters + ) -> str: + return (api_base or get_secret_str("FAL_AI_QUEUE_API_BASE") or "https://queue.fal.run").rstrip("/") + + def transform_video_create_request( + self, + model: str, + prompt: str, + api_base: str, + video_create_optional_request_params: dict[ # mutable-ok: BaseVideoConfig requires mutable parameters + str, object + ], # mutable-ok: BaseVideoConfig requires mutable parameters + litellm_params: GenericLiteLLMParams, + headers: dict[str, str], # mutable-ok: BaseVideoConfig requires mutable headers + ) -> tuple[dict[str, object], RequestFiles, str]: # mutable-ok: BaseVideoConfig requires mutable mappings + request_data: Final[dict[str, object]] = dict( # mutable-ok: HTTP JSON payload requires mutable data + MappingProxyType( + { + "prompt": prompt, + **{ # mutable-ok: dynamic request fields require a mapping + key: value for key, value in video_create_optional_request_params.items() if key != "model" + }, + } + ) + ) + return request_data, [], f"{api_base.rstrip('/')}/{model}" # mutable-ok: HTTP files payload requires a list + + def transform_video_create_response( + self, + model: str, + raw_response: httpx.Response, + logging_obj: object, + custom_llm_provider: str | None = None, + request_data: Mapping[str, object] | None = None, + ) -> VideoObject: + response_data: Final[Mapping[str, object]] = _response_data(raw_response) + request_params: Final[Mapping[str, object]] = request_data or MappingProxyType({}) + request_id: Final[str] = _response_string(response_data, "request_id") + provider: Final[str] = custom_llm_provider or _FAL_AI_PROVIDER + duration: Final[float | None] = _numeric_duration(request_params.get("duration")) + resolution: Final[object] = request_params.get("resolution") + seconds: Final[str | None] = _duration_value(request_params["duration"]) if duration is not None else None + size: Final[str | None] = resolution if isinstance(resolution, str) else None + usage: Final[dict[str, object]] = dict( # mutable-ok: VideoObject requires a mutable usage mapping + MappingProxyType( + { + key: value + for key, value in ( + ("duration_seconds", duration), + ("video_resolution", resolution if isinstance(resolution, str) else "720p"), + ) + if value is not None + } + ) + ) # mutable-ok: VideoObject requires a mutable usage mapping + video_object: Final[VideoObject] = VideoObject( + id=encode_video_id_with_provider(request_id, provider, model), + object="video", + status="queued", + created_at=int(time.time()), + model=model, + seconds=seconds, + size=size, + ) + video_object.usage = usage + return video_object + + def transform_video_status_retrieve_request( + self, + video_id: str, + api_base: str, + litellm_params: GenericLiteLLMParams, + headers: dict[str, str], # mutable-ok: BaseVideoConfig requires mutable headers + ) -> tuple[str, dict[str, object]]: # mutable-ok: BaseVideoConfig requires mutable mappings + request_id, model_id = self._decode_video_id(video_id) + encoded_request_id: Final[str] = encode_url_path_segment(request_id, field_name="video_id") + return ( + f"{api_base.rstrip('/')}/{_queue_request_base_path(model_id)}/requests/{encoded_request_id}/status", + {}, # mutable-ok: BaseVideoConfig requires a mutable mapping + ) + + def transform_video_status_retrieve_response( + self, + raw_response: httpx.Response, + logging_obj: object, + custom_llm_provider: str | None = None, + ) -> VideoObject: + response_data: Final[Mapping[str, object]] = _response_data(raw_response) + raw_status: Final[str] = _response_string(response_data, "status", "IN_QUEUE") + status: Final[str] = MappingProxyType( + { + "IN_QUEUE": "queued", + "IN_PROGRESS": "in_progress", + "COMPLETED": "completed", + } + ).get(raw_status, "queued") + error_value: Final[object] = response_data.get("error") + error: Final[str | None] = error_value if isinstance(error_value, str) else None + provider: Final[str] = custom_llm_provider or _FAL_AI_PROVIDER + return VideoObject( + id=encode_video_id_with_provider(_response_string(response_data, "request_id"), provider), + object="video", + status="failed" if error else status, + created_at=0, + error=( + {"code": "fal_error", "message": error} if error else None # mutable-ok: VideoObject requires a dict + ), # mutable-ok: VideoObject requires a dict + ) + + @staticmethod + def _decode_video_id(video_id: str) -> tuple[str, str]: + decoded: Final = decode_video_id_with_provider(video_id) + request_id: Final[str] = decoded.get("video_id", video_id) + model_id: Final[str | None] = decoded.get("model_id") + if not model_id: + raise ValueError("fal.ai video ids must be created through litellm with a model") + return request_id, model_id + + def transform_video_content_request( + self, + video_id: str, + api_base: str, + litellm_params: GenericLiteLLMParams, + headers: dict[str, str], # mutable-ok: BaseVideoConfig requires mutable headers + variant: str | None = None, + ) -> tuple[str, dict[str, str]]: # mutable-ok: BaseVideoConfig requires mutable mappings + request_id, model_id = self._decode_video_id(video_id) + encoded_request_id: Final[str] = encode_url_path_segment(request_id, field_name="video_id") + return ( + f"{api_base.rstrip('/')}/{_queue_request_base_path(model_id)}/requests/{encoded_request_id}", + {}, # mutable-ok: BaseVideoConfig requires a mutable mapping + ) + + @staticmethod + def _extract_video_url(response_data: Mapping[str, object]) -> str: + raw_video_data: Final[object] = response_data.get("video") + video_data: Final[Mapping[str, object] | None] = ( + TypeAdapter(Mapping[str, object]).validate_python(raw_video_data) + if isinstance(raw_video_data, Mapping) + else None + ) + if video_data is not None: + video_url: Final[object] = video_data.get("url") + if isinstance(video_url, str) and video_url: + return video_url + error_message: Final[str | None] = next( + (value for key in ("error", "detail") if isinstance(value := response_data.get(key), str)), + None, + ) + if error_message: + raise ValueError(f"fal.ai video result did not include a video URL: {error_message}") + raise ValueError("fal.ai video result did not include a video URL") + + def transform_video_content_response(self, raw_response: httpx.Response, logging_obj: object) -> bytes: + video_url: Final[str] = self._extract_video_url(_response_data(raw_response)) + httpx_client: Final[HTTPHandler] = _get_httpx_client() + video_response: Final[httpx.Response] = httpx_client.get( # pyright: ignore[reportUnknownMemberType] # HTTP handler stubs are untyped + video_url + ) + video_response.raise_for_status() + return video_response.content + + async def async_transform_video_content_response(self, raw_response: httpx.Response, logging_obj: object) -> bytes: + video_url: Final[str] = self._extract_video_url(_response_data(raw_response)) + async_httpx_client: Final[AsyncHTTPHandler] = get_async_httpx_client(llm_provider=LlmProviders.FAL_AI) + video_response: Final[httpx.Response] = await async_httpx_client.get( # pyright: ignore[reportUnknownMemberType] # HTTP handler stubs are untyped + video_url + ) + video_response.raise_for_status() + return video_response.content + + def transform_video_remix_request( + self, + video_id: str, + prompt: str, + api_base: str, + litellm_params: GenericLiteLLMParams, + headers: dict[str, str], # mutable-ok: BaseVideoConfig requires mutable headers + extra_body: Mapping[str, object] | None = None, + ) -> tuple[str, dict[str, object]]: # mutable-ok: BaseVideoConfig requires mutable mappings + raise NotImplementedError("video remix is not supported for fal.ai") + + def transform_video_remix_response( + self, + raw_response: httpx.Response, + logging_obj: object, + custom_llm_provider: str | None = None, + ) -> VideoObject: + raise NotImplementedError("video remix is not supported for fal.ai") + + def transform_video_list_request( + self, + api_base: str, + litellm_params: GenericLiteLLMParams, + headers: dict[str, str], # mutable-ok: BaseVideoConfig requires mutable headers + after: str | None = None, + limit: int | None = None, + order: str | None = None, + extra_query: Mapping[str, object] | None = None, + ) -> tuple[str, dict[str, object]]: # mutable-ok: BaseVideoConfig requires mutable mappings + raise NotImplementedError("video listing is not supported for fal.ai") + + def transform_video_list_response( + self, + raw_response: httpx.Response, + logging_obj: object, + custom_llm_provider: str | None = None, + ) -> dict[str, str]: # mutable-ok: BaseVideoConfig requires mutable mappings + raise NotImplementedError("video listing is not supported for fal.ai") + + def transform_video_delete_request( + self, + video_id: str, + api_base: str, + litellm_params: GenericLiteLLMParams, + headers: dict[str, str], # mutable-ok: BaseVideoConfig requires mutable headers + ) -> tuple[str, dict[str, object]]: # mutable-ok: BaseVideoConfig requires mutable mappings + raise NotImplementedError("video delete is not supported for fal.ai") + + def transform_video_delete_response(self, raw_response: httpx.Response, logging_obj: object) -> VideoObject: + raise NotImplementedError("video delete is not supported for fal.ai") + + def transform_video_create_character_request( + self, + name: str, + video: object, + api_base: str, + litellm_params: GenericLiteLLMParams, + headers: dict[str, str], # mutable-ok: BaseVideoConfig requires mutable headers + ) -> tuple[str, list[object]]: # mutable-ok: BaseVideoConfig requires mutable lists + raise NotImplementedError("video character creation is not supported for fal.ai") + + def transform_video_create_character_response( + self, + raw_response: httpx.Response, + logging_obj: object, + ) -> CharacterObject: + raise NotImplementedError("video character creation is not supported for fal.ai") + + def transform_video_get_character_request( + self, + character_id: str, + api_base: str, + litellm_params: GenericLiteLLMParams, + headers: dict[str, str], # mutable-ok: BaseVideoConfig requires mutable headers + ) -> tuple[str, dict[str, object]]: # mutable-ok: BaseVideoConfig requires mutable mappings + raise NotImplementedError("video character retrieval is not supported for fal.ai") + + def transform_video_get_character_response( + self, + raw_response: httpx.Response, + logging_obj: object, + ) -> CharacterObject: + raise NotImplementedError("video character retrieval is not supported for fal.ai") + + def transform_video_edit_request( + self, + prompt: str, + video_id: str, + api_base: str, + litellm_params: GenericLiteLLMParams, + headers: dict[str, str], # mutable-ok: BaseVideoConfig requires mutable headers + video_file: FileContent | None = None, + extra_body: Mapping[str, object] | None = None, + prefetched_source_data: Mapping[str, object] | None = None, + ) -> tuple[str, Mapping[str, object], RequestFiles | None]: + raise NotImplementedError("video edit is not supported for fal.ai") + + def transform_video_edit_response( + self, + raw_response: httpx.Response, + logging_obj: object, + custom_llm_provider: str | None = None, + request_data: Mapping[str, object] | None = None, + ) -> VideoObject: + raise NotImplementedError("video edit is not supported for fal.ai") + + def transform_video_extension_request( + self, + prompt: str, + video_id: str, + seconds: str, + api_base: str, + litellm_params: GenericLiteLLMParams, + headers: dict[str, str], # mutable-ok: BaseVideoConfig requires mutable headers + extra_body: Mapping[str, object] | None = None, + ) -> tuple[str, dict[str, object]]: # mutable-ok: BaseVideoConfig requires mutable mappings + raise NotImplementedError("video extension is not supported for fal.ai") + + def transform_video_extension_response( + self, + raw_response: httpx.Response, + logging_obj: object, + custom_llm_provider: str | None = None, + ) -> VideoObject: + raise NotImplementedError("video extension is not supported for fal.ai") + + def get_error_class( + self, + error_message: str, + status_code: int, + headers: dict[str, str] | httpx.Headers, # mutable-ok: BaseLLMException requires mutable headers + ) -> BaseLLMException: + return FalAIVideoError(status_code=status_code, message=error_message, headers=headers) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 4dbf0337894..324eb6b2d66 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -22807,6 +22807,127 @@ "/v1/images/generations" ] }, + "fal_ai/bytedance/seedance-2.5/text-to-video": { + "litellm_provider": "fal_ai", + "mode": "video_generation", + "output_cost_per_second": 0.473, + "output_cost_per_second_480p": 0.2205, + "output_cost_per_second_720p": 0.473, + "source": "https://fal.ai/models/bytedance/seedance-2.5/text-to-video", + "supported_endpoints": [ + "/v1/videos" + ], + "supported_modalities": [ + "text" + ], + "supported_output_modalities": [ + "video" + ] + }, + "fal_ai/bytedance/seedance-2.5/image-to-video": { + "litellm_provider": "fal_ai", + "mode": "video_generation", + "output_cost_per_second": 0.473, + "output_cost_per_second_480p": 0.2205, + "output_cost_per_second_720p": 0.473, + "source": "https://fal.ai/models/bytedance/seedance-2.5/image-to-video", + "supported_endpoints": [ + "/v1/videos" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ] + }, + "fal_ai/bytedance/seedance-2.5/reference-to-video": { + "litellm_provider": "fal_ai", + "mode": "video_generation", + "output_cost_per_second": 0.473, + "output_cost_per_second_480p": 0.2205, + "output_cost_per_second_720p": 0.473, + "source": "https://fal.ai/models/bytedance/seedance-2.5/reference-to-video", + "supported_endpoints": [ + "/v1/videos" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ] + }, + "fal_ai/bytedance/seedance-2.0/text-to-video": { + "litellm_provider": "fal_ai", + "mode": "video_generation", + "output_cost_per_second": 0.3034, + "output_cost_per_second_480p": 0.1346, + "output_cost_per_second_720p": 0.3034, + "output_cost_per_second_1080p": 0.682, + "output_cost_per_second_4k": 1.5552, + "source": "https://fal.ai/models/bytedance/seedance-2.0/text-to-video", + "metadata": { + "comment": "fal bills $0.014 per 1k tokens (480p/720p/1080p) and $0.008 per 1k tokens (4k) with tokens = h*w*seconds*24/1024; 480p and 4k rates derived from that formula at 854x480 and 3840x2160" + }, + "supported_endpoints": [ + "/v1/videos" + ], + "supported_modalities": [ + "text" + ], + "supported_output_modalities": [ + "video" + ] + }, + "fal_ai/bytedance/seedance-2.0/image-to-video": { + "litellm_provider": "fal_ai", + "mode": "video_generation", + "output_cost_per_second": 0.3034, + "output_cost_per_second_480p": 0.1346, + "output_cost_per_second_720p": 0.3034, + "output_cost_per_second_1080p": 0.682, + "output_cost_per_second_4k": 1.5552, + "source": "https://fal.ai/models/bytedance/seedance-2.0/image-to-video", + "metadata": { + "comment": "fal bills $0.014 per 1k tokens (480p/720p/1080p) and $0.008 per 1k tokens (4k) with tokens = h*w*seconds*24/1024; 480p and 4k rates derived from that formula at 854x480 and 3840x2160" + }, + "supported_endpoints": [ + "/v1/videos" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ] + }, + "fal_ai/bytedance/seedance-2.0/reference-to-video": { + "litellm_provider": "fal_ai", + "mode": "video_generation", + "output_cost_per_second": 0.3034, + "output_cost_per_second_480p": 0.1346, + "output_cost_per_second_720p": 0.3034, + "output_cost_per_second_1080p": 0.682, + "output_cost_per_second_4k": 1.5552, + "source": "https://fal.ai/models/bytedance/seedance-2.0/reference-to-video", + "metadata": { + "comment": "fal bills $0.014 per 1k tokens (480p/720p/1080p) and $0.008 per 1k tokens (4k) with tokens = h*w*seconds*24/1024; 480p and 4k rates derived from that formula at 854x480 and 3840x2160" + }, + "supported_endpoints": [ + "/v1/videos" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ] + }, "fal_ai/fal-ai/ideogram/v3": { "litellm_provider": "fal_ai", "mode": "image_generation", diff --git a/litellm/utils.py b/litellm/utils.py index 48d13bc16af..3991cecdac6 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -9403,6 +9403,10 @@ class ProviderConfigManager: from litellm.llms.runwayml.videos.transformation import RunwayMLVideoConfig return RunwayMLVideoConfig() + elif LlmProviders.FAL_AI == provider: + from litellm.llms.fal_ai.videos.transformation import FalAIVideoConfig + + return FalAIVideoConfig() elif LlmProviders.HOSTED_VLLM == provider: from litellm.llms.hosted_vllm.videos import get_hosted_vllm_video_config diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 4dbf0337894..324eb6b2d66 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -22807,6 +22807,127 @@ "/v1/images/generations" ] }, + "fal_ai/bytedance/seedance-2.5/text-to-video": { + "litellm_provider": "fal_ai", + "mode": "video_generation", + "output_cost_per_second": 0.473, + "output_cost_per_second_480p": 0.2205, + "output_cost_per_second_720p": 0.473, + "source": "https://fal.ai/models/bytedance/seedance-2.5/text-to-video", + "supported_endpoints": [ + "/v1/videos" + ], + "supported_modalities": [ + "text" + ], + "supported_output_modalities": [ + "video" + ] + }, + "fal_ai/bytedance/seedance-2.5/image-to-video": { + "litellm_provider": "fal_ai", + "mode": "video_generation", + "output_cost_per_second": 0.473, + "output_cost_per_second_480p": 0.2205, + "output_cost_per_second_720p": 0.473, + "source": "https://fal.ai/models/bytedance/seedance-2.5/image-to-video", + "supported_endpoints": [ + "/v1/videos" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ] + }, + "fal_ai/bytedance/seedance-2.5/reference-to-video": { + "litellm_provider": "fal_ai", + "mode": "video_generation", + "output_cost_per_second": 0.473, + "output_cost_per_second_480p": 0.2205, + "output_cost_per_second_720p": 0.473, + "source": "https://fal.ai/models/bytedance/seedance-2.5/reference-to-video", + "supported_endpoints": [ + "/v1/videos" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ] + }, + "fal_ai/bytedance/seedance-2.0/text-to-video": { + "litellm_provider": "fal_ai", + "mode": "video_generation", + "output_cost_per_second": 0.3034, + "output_cost_per_second_480p": 0.1346, + "output_cost_per_second_720p": 0.3034, + "output_cost_per_second_1080p": 0.682, + "output_cost_per_second_4k": 1.5552, + "source": "https://fal.ai/models/bytedance/seedance-2.0/text-to-video", + "metadata": { + "comment": "fal bills $0.014 per 1k tokens (480p/720p/1080p) and $0.008 per 1k tokens (4k) with tokens = h*w*seconds*24/1024; 480p and 4k rates derived from that formula at 854x480 and 3840x2160" + }, + "supported_endpoints": [ + "/v1/videos" + ], + "supported_modalities": [ + "text" + ], + "supported_output_modalities": [ + "video" + ] + }, + "fal_ai/bytedance/seedance-2.0/image-to-video": { + "litellm_provider": "fal_ai", + "mode": "video_generation", + "output_cost_per_second": 0.3034, + "output_cost_per_second_480p": 0.1346, + "output_cost_per_second_720p": 0.3034, + "output_cost_per_second_1080p": 0.682, + "output_cost_per_second_4k": 1.5552, + "source": "https://fal.ai/models/bytedance/seedance-2.0/image-to-video", + "metadata": { + "comment": "fal bills $0.014 per 1k tokens (480p/720p/1080p) and $0.008 per 1k tokens (4k) with tokens = h*w*seconds*24/1024; 480p and 4k rates derived from that formula at 854x480 and 3840x2160" + }, + "supported_endpoints": [ + "/v1/videos" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ] + }, + "fal_ai/bytedance/seedance-2.0/reference-to-video": { + "litellm_provider": "fal_ai", + "mode": "video_generation", + "output_cost_per_second": 0.3034, + "output_cost_per_second_480p": 0.1346, + "output_cost_per_second_720p": 0.3034, + "output_cost_per_second_1080p": 0.682, + "output_cost_per_second_4k": 1.5552, + "source": "https://fal.ai/models/bytedance/seedance-2.0/reference-to-video", + "metadata": { + "comment": "fal bills $0.014 per 1k tokens (480p/720p/1080p) and $0.008 per 1k tokens (4k) with tokens = h*w*seconds*24/1024; 480p and 4k rates derived from that formula at 854x480 and 3840x2160" + }, + "supported_endpoints": [ + "/v1/videos" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ] + }, "fal_ai/fal-ai/ideogram/v3": { "litellm_provider": "fal_ai", "mode": "image_generation", diff --git a/tests/test_litellm/llms/fal_ai/videos/test_fal_ai_video_transformation.py b/tests/test_litellm/llms/fal_ai/videos/test_fal_ai_video_transformation.py new file mode 100644 index 00000000000..d4b058ddcf6 --- /dev/null +++ b/tests/test_litellm/llms/fal_ai/videos/test_fal_ai_video_transformation.py @@ -0,0 +1,231 @@ +from unittest.mock import Mock + +import httpx +import pytest + +import litellm +import litellm.llms.fal_ai.videos.transformation as fal_video_module +from litellm.cost_calculator import default_video_cost_calculator +from litellm.llms.fal_ai.videos.transformation import ( + FalAIVideoConfig, + FalAIVideoError, + _queue_request_base_path, +) +from litellm.types.router import GenericLiteLLMParams +from litellm.types.utils import LlmProviders +from litellm.types.videos.utils import decode_video_id_with_provider +from litellm.utils import ProviderConfigManager + +MODEL = "bytedance/seedance-2.5/text-to-video" + + +class TestFalAIVideoTransformation: + def setup_method(self): + self.config = FalAIVideoConfig() + self.logging_obj = Mock() + + def test_map_openai_params(self): + mapped = self.config.map_openai_params( + { + "seconds": "5", + "size": "1280x720", + "input_reference": "https://example.com/image.png", + "user": "user-123", + "generate_audio": False, + }, + MODEL, + False, + ) + + assert mapped == { + "duration": "5", + "resolution": "720p", + "aspect_ratio": "16:9", + "image_url": "https://example.com/image.png", + "end_user_id": "user-123", + "generate_audio": False, + } + + assert self.config.map_openai_params({"size": "1080x1080"}, MODEL, False) == { + "resolution": "1080p", + "aspect_ratio": "1:1", + } + assert self.config.map_openai_params({"size": "720p"}, MODEL, False) == {"resolution": "720p"} + + def test_map_openai_params_rejects_non_url_input_reference(self): + with pytest.raises(ValueError, match="public image URL"): + self.config.map_openai_params({"input_reference": b"image"}, MODEL, False) + + def test_transform_video_create_request(self): + body, files, url = self.config.transform_video_create_request( + model=MODEL, + prompt="A quiet ocean at sunrise", + api_base="https://queue.fal.run", + video_create_optional_request_params={ + "duration": "5", + "resolution": "480p", + "aspect_ratio": "16:9", + "generate_audio": False, + "model": MODEL, + }, + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + assert url == f"https://queue.fal.run/{MODEL}" + assert files == [] + assert body == { + "prompt": "A quiet ocean at sunrise", + "duration": "5", + "resolution": "480p", + "aspect_ratio": "16:9", + "generate_audio": False, + } + assert "model" not in body + + def test_transform_video_create_response_encodes_model_and_usage(self): + response = Mock(spec=httpx.Response) + response.json.return_value = {"request_id": "abc"} + + video = self.config.transform_video_create_response( + model=MODEL, + raw_response=response, + logging_obj=self.logging_obj, + custom_llm_provider="fal_ai", + request_data={"duration": "5", "resolution": "480p"}, + ) + + decoded = decode_video_id_with_provider(video.id) + assert decoded["custom_llm_provider"] == "fal_ai" + assert decoded["model_id"] == MODEL + assert decoded["video_id"] == "abc" + assert video.status == "queued" + assert video.usage == {"duration_seconds": 5.0, "video_resolution": "480p"} + + auto_video = self.config.transform_video_create_response( + model=MODEL, + raw_response=response, + logging_obj=self.logging_obj, + custom_llm_provider="fal_ai", + request_data={"duration": "auto"}, + ) + assert auto_video.usage == {"video_resolution": "720p"} + assert auto_video.seconds is None + assert auto_video.size is None + + def test_status_request_uses_queue_base_path(self): + response = Mock(spec=httpx.Response) + response.json.return_value = {"request_id": "abc"} + video = self.config.transform_video_create_response( + model=MODEL, + raw_response=response, + logging_obj=self.logging_obj, + custom_llm_provider="fal_ai", + request_data={}, + ) + + url, params = self.config.transform_video_status_retrieve_request( + video_id=video.id, + api_base="https://queue.fal.run", + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + assert url == "https://queue.fal.run/bytedance/seedance-2.5/requests/abc/status" + assert params == {} + assert _queue_request_base_path("workflows/owner/app/x") == "workflows/owner/app" + assert _queue_request_base_path("comfy/owner/app/x") == "comfy/owner/app" + + def test_status_request_rejects_unencoded_video_id(self): + with pytest.raises(ValueError, match="must be created through litellm"): + self.config.transform_video_status_retrieve_request( + video_id="abc", + api_base="https://queue.fal.run", + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + @pytest.mark.parametrize( + ("response_data", "expected_status"), + [ + ({"request_id": "abc", "status": "IN_QUEUE"}, "queued"), + ({"request_id": "abc", "status": "IN_PROGRESS"}, "in_progress"), + ({"request_id": "abc", "status": "COMPLETED"}, "completed"), + ], + ) + def test_status_response_mapping(self, response_data, expected_status): + response = Mock(spec=httpx.Response) + response.json.return_value = response_data + + video = self.config.transform_video_status_retrieve_response( + raw_response=response, + logging_obj=self.logging_obj, + custom_llm_provider="fal_ai", + ) + + assert video.status == expected_status + assert video.created_at == 0 + + def test_status_response_error(self): + response = Mock(spec=httpx.Response) + response.json.return_value = { + "request_id": "abc", + "status": "COMPLETED", + "error": "generation failed", + } + + video = self.config.transform_video_status_retrieve_response( + raw_response=response, + logging_obj=self.logging_obj, + custom_llm_provider="fal_ai", + ) + + assert video.status == "failed" + assert video.error == {"code": "fal_error", "message": "generation failed"} + + def test_content_response_downloads_video_url(self, monkeypatch): + content_response = httpx.Response( + 200, + content=b"video-bytes", + request=httpx.Request("GET", "https://cdn.example.com/video.mp4"), + ) + + class FakeHTTPClient: + def get(self, url): + assert url == "https://cdn.example.com/video.mp4" + return content_response + + monkeypatch.setattr(fal_video_module, "_get_httpx_client", lambda: FakeHTTPClient()) + response = Mock(spec=httpx.Response) + response.json.return_value = {"video": {"url": "https://cdn.example.com/video.mp4"}} + + assert self.config.transform_video_content_response(response, self.logging_obj) == b"video-bytes" + + def test_content_response_rejects_missing_video(self): + response = Mock(spec=httpx.Response) + response.json.return_value = {"error": "generation failed"} + + with pytest.raises(ValueError, match="generation failed"): + self.config.transform_video_content_response(response, self.logging_obj) + + def test_provider_config_and_error_class(self): + provider_config = ProviderConfigManager.get_provider_video_config( + model=MODEL, + provider=LlmProviders.FAL_AI, + ) + assert isinstance(provider_config, FalAIVideoConfig) + assert isinstance(self.config.get_error_class("bad key", 401, {}), FalAIVideoError) + + def test_video_cost_uses_tiered_rows(self): + rows = { + model: row + for model, row in litellm.model_cost.items() + if row.get("litellm_provider") == "fal_ai" and row.get("mode") == "video_generation" + } + assert rows + for model, row in rows.items(): + assert default_video_cost_calculator(model, 5, "fal_ai", video_resolution="480p") == ( + 5 * row["output_cost_per_second_480p"] + ) + assert default_video_cost_calculator(model, 5, "fal_ai", video_resolution="720p") == ( + 5 * row["output_cost_per_second"] + ) From 0f4ce95492cde57b1f2eb29a1ea119dda0b13e24 Mon Sep 17 00:00:00 2001 From: kerry Date: Sat, 19 Sep 2026 16:45:53 +0000 Subject: [PATCH 2/9] refactor(fal_ai): simplify video config mappings Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/llms/fal_ai/videos/transformation.py | 178 +++++++++---------- 1 file changed, 80 insertions(+), 98 deletions(-) diff --git a/litellm/llms/fal_ai/videos/transformation.py b/litellm/llms/fal_ai/videos/transformation.py index f8ebf828d68..f0142ddc3de 100644 --- a/litellm/llms/fal_ai/videos/transformation.py +++ b/litellm/llms/fal_ai/videos/transformation.py @@ -2,7 +2,7 @@ import math import time from collections.abc import Mapping from types import MappingProxyType -from typing import Final +from typing import Final, TypeAlias import httpx from httpx._types import FileContent, RequestFiles @@ -42,12 +42,25 @@ _RESOLUTION_TIERS: Final[tuple[tuple[int, str], ...]] = ( (720, "720p"), (1080, "1080p"), ) +_QUEUE_NAMESPACES: Final[frozenset[str]] = frozenset(("workflows", "comfy")) +_STATUS_MAP: Final[Mapping[str, str]] = MappingProxyType( + { + "IN_QUEUE": "queued", + "IN_PROGRESS": "in_progress", + "COMPLETED": "completed", + } +) _FAL_AI_PROVIDER: Final[str] = LlmProviders.FAL_AI.value +_SupportedParams: TypeAlias = list[str] +_VideoParams: TypeAlias = dict[str, object] +_VideoHeaders: TypeAlias = dict[str, str] +_VideoStringParams: TypeAlias = dict[str, str] +_VideoFiles: TypeAlias = list[object] def _queue_request_base_path(model: str) -> str: segments: Final[tuple[str, ...]] = tuple(model.split("/")) - segment_count: Final[int] = 3 if segments and segments[0] in frozenset(("workflows", "comfy")) else 2 + segment_count: Final[int] = 3 if segments and segments[0] in _QUEUE_NAMESPACES else 2 return "/".join(segments[:segment_count]) @@ -105,8 +118,8 @@ def _response_string(response_data: Mapping[str, object], key: str, default: str class FalAIVideoConfig(BaseVideoConfig): - def get_supported_openai_params(self, model: str) -> list[str]: # mutable-ok: API contract requires a list - return [ # mutable-ok: API contract requires a list + def get_supported_openai_params(self, model: str) -> _SupportedParams: + supported_params: Final[_SupportedParams] = [ # mutable-ok: BaseVideoConfig requires a list "model", "prompt", "input_reference", @@ -115,23 +128,22 @@ class FalAIVideoConfig(BaseVideoConfig): "user", "extra_headers", ] + return supported_params def map_openai_params( self, video_create_optional_params: VideoCreateOptionalRequestParams, model: str, drop_params: bool, - ) -> dict[str, object]: # mutable-ok: BaseVideoConfig requires a mutable mapping + ) -> _VideoParams: supported_params: Final[frozenset[str]] = frozenset(self.get_supported_openai_params(model)) input_reference: Final[object] = video_create_optional_params.get("input_reference") + if "input_reference" in video_create_optional_params and not isinstance(input_reference, str): + raise ValueError("fal.ai needs a public image URL for input_reference") input_reference_params: Final[Mapping[str, str]] = ( MappingProxyType({}) - if "input_reference" not in video_create_optional_params - else ( - MappingProxyType({"image_url": input_reference}) - if isinstance(input_reference, str) - else self._invalid_input_reference() - ) + if not isinstance(input_reference, str) + else MappingProxyType({"image_url": input_reference}) ) duration_params: Final[Mapping[str, str]] = ( MappingProxyType({}) @@ -139,7 +151,7 @@ class FalAIVideoConfig(BaseVideoConfig): else self._duration_params(video_create_optional_params["seconds"]) ) size_params: Final[Mapping[str, str]] = ( - self._size_params(video_create_optional_params["size"]) + _size_params(video_create_optional_params["size"]) if "size" in video_create_optional_params else MappingProxyType({}) ) @@ -148,23 +160,16 @@ class FalAIVideoConfig(BaseVideoConfig): if isinstance(user := video_create_optional_params.get("user"), str) else MappingProxyType({}) ) - return dict( # mutable-ok: BaseVideoConfig requires a mutable mapping - MappingProxyType( - { - **input_reference_params, - **duration_params, - **size_params, - **user_params, - **{ # mutable-ok: dynamic passthrough fields require a mapping - key: value for key, value in video_create_optional_params.items() if key not in supported_params - }, - } - ) - ) # mutable-ok: BaseVideoConfig requires a mutable mapping - - @staticmethod - def _invalid_input_reference() -> Mapping[str, str]: - raise ValueError("fal.ai needs a public image URL for input_reference") + mapped_params: Final[_VideoParams] = { + **input_reference_params, + **duration_params, + **size_params, + **user_params, + **{ # mutable-ok: BaseVideoConfig requires a mutable parameter mapping + key: value for key, value in video_create_optional_params.items() if key not in supported_params + }, + } + return mapped_params @staticmethod def _duration_params(seconds: object) -> Mapping[str, str]: @@ -173,17 +178,13 @@ class FalAIVideoConfig(BaseVideoConfig): raise ValueError("fal.ai seconds must be a numeric value") return MappingProxyType({"duration": duration}) - @staticmethod - def _size_params(size: object) -> Mapping[str, str]: - return _size_params(size) - def validate_environment( self, - headers: dict[str, str], # mutable-ok: BaseVideoConfig requires mutable headers + headers: _VideoHeaders, model: str, api_key: str | None = None, litellm_params: GenericLiteLLMParams | None = None, - ) -> dict[str, str]: # mutable-ok: BaseVideoConfig requires mutable headers + ) -> _VideoHeaders: final_api_key: Final[str | None] = ( api_key or (litellm_params.api_key if litellm_params is not None else None) @@ -192,21 +193,18 @@ class FalAIVideoConfig(BaseVideoConfig): ) if not final_api_key: raise ValueError("fal.ai API key is required") - return dict( # mutable-ok: BaseVideoConfig requires mutable headers - MappingProxyType( - { - **headers, - "Authorization": f"Key {final_api_key}", - "Content-Type": "application/json", - } - ) - ) # mutable-ok: BaseVideoConfig requires mutable headers + validated_headers: Final[_VideoHeaders] = { + **headers, + "Authorization": f"Key {final_api_key}", + "Content-Type": "application/json", + } + return validated_headers def get_complete_url( self, model: str, api_base: str | None, - litellm_params: dict[str, object], # mutable-ok: BaseVideoConfig requires mutable parameters + litellm_params: _VideoParams, ) -> str: return (api_base or get_secret_str("FAL_AI_QUEUE_API_BASE") or "https://queue.fal.run").rstrip("/") @@ -215,22 +213,16 @@ class FalAIVideoConfig(BaseVideoConfig): model: str, prompt: str, api_base: str, - video_create_optional_request_params: dict[ # mutable-ok: BaseVideoConfig requires mutable parameters - str, object - ], # mutable-ok: BaseVideoConfig requires mutable parameters + video_create_optional_request_params: _VideoParams, litellm_params: GenericLiteLLMParams, - headers: dict[str, str], # mutable-ok: BaseVideoConfig requires mutable headers - ) -> tuple[dict[str, object], RequestFiles, str]: # mutable-ok: BaseVideoConfig requires mutable mappings - request_data: Final[dict[str, object]] = dict( # mutable-ok: HTTP JSON payload requires mutable data - MappingProxyType( - { - "prompt": prompt, - **{ # mutable-ok: dynamic request fields require a mapping - key: value for key, value in video_create_optional_request_params.items() if key != "model" - }, - } - ) - ) + headers: _VideoHeaders, + ) -> tuple[_VideoParams, RequestFiles, str]: + request_data: Final[_VideoParams] = { + "prompt": prompt, + **{ # mutable-ok: HTTP JSON payload requires a mutable mapping + key: value for key, value in video_create_optional_request_params.items() if key != "model" + }, + } return request_data, [], f"{api_base.rstrip('/')}/{model}" # mutable-ok: HTTP files payload requires a list def transform_video_create_response( @@ -249,18 +241,14 @@ class FalAIVideoConfig(BaseVideoConfig): resolution: Final[object] = request_params.get("resolution") seconds: Final[str | None] = _duration_value(request_params["duration"]) if duration is not None else None size: Final[str | None] = resolution if isinstance(resolution, str) else None - usage: Final[dict[str, object]] = dict( # mutable-ok: VideoObject requires a mutable usage mapping - MappingProxyType( - { - key: value - for key, value in ( - ("duration_seconds", duration), - ("video_resolution", resolution if isinstance(resolution, str) else "720p"), - ) - if value is not None - } + usage: Final[_VideoParams] = { # mutable-ok: VideoObject requires a mutable usage mapping + key: value + for key, value in ( + ("duration_seconds", duration), + ("video_resolution", resolution if isinstance(resolution, str) else "720p"), ) - ) # mutable-ok: VideoObject requires a mutable usage mapping + if value is not None + } video_object: Final[VideoObject] = VideoObject( id=encode_video_id_with_provider(request_id, provider, model), object="video", @@ -278,8 +266,8 @@ class FalAIVideoConfig(BaseVideoConfig): video_id: str, api_base: str, litellm_params: GenericLiteLLMParams, - headers: dict[str, str], # mutable-ok: BaseVideoConfig requires mutable headers - ) -> tuple[str, dict[str, object]]: # mutable-ok: BaseVideoConfig requires mutable mappings + headers: _VideoHeaders, + ) -> tuple[str, _VideoParams]: request_id, model_id = self._decode_video_id(video_id) encoded_request_id: Final[str] = encode_url_path_segment(request_id, field_name="video_id") return ( @@ -295,13 +283,7 @@ class FalAIVideoConfig(BaseVideoConfig): ) -> VideoObject: response_data: Final[Mapping[str, object]] = _response_data(raw_response) raw_status: Final[str] = _response_string(response_data, "status", "IN_QUEUE") - status: Final[str] = MappingProxyType( - { - "IN_QUEUE": "queued", - "IN_PROGRESS": "in_progress", - "COMPLETED": "completed", - } - ).get(raw_status, "queued") + status: Final[str] = _STATUS_MAP.get(raw_status, "queued") error_value: Final[object] = response_data.get("error") error: Final[str | None] = error_value if isinstance(error_value, str) else None provider: Final[str] = custom_llm_provider or _FAL_AI_PROVIDER @@ -312,7 +294,7 @@ class FalAIVideoConfig(BaseVideoConfig): created_at=0, error=( {"code": "fal_error", "message": error} if error else None # mutable-ok: VideoObject requires a dict - ), # mutable-ok: VideoObject requires a dict + ), ) @staticmethod @@ -329,9 +311,9 @@ class FalAIVideoConfig(BaseVideoConfig): video_id: str, api_base: str, litellm_params: GenericLiteLLMParams, - headers: dict[str, str], # mutable-ok: BaseVideoConfig requires mutable headers + headers: _VideoHeaders, variant: str | None = None, - ) -> tuple[str, dict[str, str]]: # mutable-ok: BaseVideoConfig requires mutable mappings + ) -> tuple[str, _VideoStringParams]: request_id, model_id = self._decode_video_id(video_id) encoded_request_id: Final[str] = encode_url_path_segment(request_id, field_name="video_id") return ( @@ -383,9 +365,9 @@ class FalAIVideoConfig(BaseVideoConfig): prompt: str, api_base: str, litellm_params: GenericLiteLLMParams, - headers: dict[str, str], # mutable-ok: BaseVideoConfig requires mutable headers + headers: _VideoHeaders, extra_body: Mapping[str, object] | None = None, - ) -> tuple[str, dict[str, object]]: # mutable-ok: BaseVideoConfig requires mutable mappings + ) -> tuple[str, _VideoParams]: raise NotImplementedError("video remix is not supported for fal.ai") def transform_video_remix_response( @@ -400,12 +382,12 @@ class FalAIVideoConfig(BaseVideoConfig): self, api_base: str, litellm_params: GenericLiteLLMParams, - headers: dict[str, str], # mutable-ok: BaseVideoConfig requires mutable headers + headers: _VideoHeaders, after: str | None = None, limit: int | None = None, order: str | None = None, extra_query: Mapping[str, object] | None = None, - ) -> tuple[str, dict[str, object]]: # mutable-ok: BaseVideoConfig requires mutable mappings + ) -> tuple[str, _VideoParams]: raise NotImplementedError("video listing is not supported for fal.ai") def transform_video_list_response( @@ -413,7 +395,7 @@ class FalAIVideoConfig(BaseVideoConfig): raw_response: httpx.Response, logging_obj: object, custom_llm_provider: str | None = None, - ) -> dict[str, str]: # mutable-ok: BaseVideoConfig requires mutable mappings + ) -> _VideoStringParams: raise NotImplementedError("video listing is not supported for fal.ai") def transform_video_delete_request( @@ -421,8 +403,8 @@ class FalAIVideoConfig(BaseVideoConfig): video_id: str, api_base: str, litellm_params: GenericLiteLLMParams, - headers: dict[str, str], # mutable-ok: BaseVideoConfig requires mutable headers - ) -> tuple[str, dict[str, object]]: # mutable-ok: BaseVideoConfig requires mutable mappings + headers: _VideoHeaders, + ) -> tuple[str, _VideoParams]: raise NotImplementedError("video delete is not supported for fal.ai") def transform_video_delete_response(self, raw_response: httpx.Response, logging_obj: object) -> VideoObject: @@ -434,8 +416,8 @@ class FalAIVideoConfig(BaseVideoConfig): video: object, api_base: str, litellm_params: GenericLiteLLMParams, - headers: dict[str, str], # mutable-ok: BaseVideoConfig requires mutable headers - ) -> tuple[str, list[object]]: # mutable-ok: BaseVideoConfig requires mutable lists + headers: _VideoHeaders, + ) -> tuple[str, _VideoFiles]: raise NotImplementedError("video character creation is not supported for fal.ai") def transform_video_create_character_response( @@ -450,8 +432,8 @@ class FalAIVideoConfig(BaseVideoConfig): character_id: str, api_base: str, litellm_params: GenericLiteLLMParams, - headers: dict[str, str], # mutable-ok: BaseVideoConfig requires mutable headers - ) -> tuple[str, dict[str, object]]: # mutable-ok: BaseVideoConfig requires mutable mappings + headers: _VideoHeaders, + ) -> tuple[str, _VideoParams]: raise NotImplementedError("video character retrieval is not supported for fal.ai") def transform_video_get_character_response( @@ -467,7 +449,7 @@ class FalAIVideoConfig(BaseVideoConfig): video_id: str, api_base: str, litellm_params: GenericLiteLLMParams, - headers: dict[str, str], # mutable-ok: BaseVideoConfig requires mutable headers + headers: _VideoHeaders, video_file: FileContent | None = None, extra_body: Mapping[str, object] | None = None, prefetched_source_data: Mapping[str, object] | None = None, @@ -490,9 +472,9 @@ class FalAIVideoConfig(BaseVideoConfig): seconds: str, api_base: str, litellm_params: GenericLiteLLMParams, - headers: dict[str, str], # mutable-ok: BaseVideoConfig requires mutable headers + headers: _VideoHeaders, extra_body: Mapping[str, object] | None = None, - ) -> tuple[str, dict[str, object]]: # mutable-ok: BaseVideoConfig requires mutable mappings + ) -> tuple[str, _VideoParams]: raise NotImplementedError("video extension is not supported for fal.ai") def transform_video_extension_response( @@ -507,6 +489,6 @@ class FalAIVideoConfig(BaseVideoConfig): self, error_message: str, status_code: int, - headers: dict[str, str] | httpx.Headers, # mutable-ok: BaseLLMException requires mutable headers + headers: _VideoHeaders | httpx.Headers, ) -> BaseLLMException: return FalAIVideoError(status_code=status_code, message=error_message, headers=headers) From 141548dcf3ec7588e86c2e304fca33be6c4cdc7e Mon Sep 17 00:00:00 2001 From: kerry Date: Sat, 19 Sep 2026 16:54:11 +0000 Subject: [PATCH 3/9] fix(fal_ai): keep status ids pollable and size resolution by the short side Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/llms/fal_ai/videos/transformation.py | 17 +++++++-- .../test_fal_ai_video_transformation.py | 37 +++++++++++++++++++ 2 files changed, 50 insertions(+), 4 deletions(-) diff --git a/litellm/llms/fal_ai/videos/transformation.py b/litellm/llms/fal_ai/videos/transformation.py index f0142ddc3de..4ca1f918c90 100644 --- a/litellm/llms/fal_ai/videos/transformation.py +++ b/litellm/llms/fal_ai/videos/transformation.py @@ -75,8 +75,16 @@ def _duration_value(value: object) -> str | None: return None -def _resolution_for_height(height: int) -> str: - return next((resolution for threshold, resolution in _RESOLUTION_TIERS if height <= threshold), "4k") +def _resolution_for_short_side(short_side: int) -> str: + return next((resolution for threshold, resolution in _RESOLUTION_TIERS if short_side <= threshold), "4k") + + +def _model_path_from_queue_url(url: object) -> str | None: + if not isinstance(url, str) or not url: + return None + path: Final[str] = httpx.URL(url).path.strip("/") + model_path, separator, _ = path.partition("/requests/") + return model_path if separator and model_path else None def _size_params(size: object) -> Mapping[str, str]: @@ -95,7 +103,7 @@ def _size_params(size: object) -> Mapping[str, str]: return MappingProxyType({}) reduced_gcd: Final[int] = math.gcd(width, height) aspect_ratio: Final[str] = f"{width // reduced_gcd}:{height // reduced_gcd}" - resolution: Final[str] = _resolution_for_height(height) + resolution: Final[str] = _resolution_for_short_side(min(width, height)) if aspect_ratio in _ALLOWED_ASPECT_RATIOS: return MappingProxyType({"resolution": resolution, "aspect_ratio": aspect_ratio}) return MappingProxyType({"resolution": resolution}) @@ -287,8 +295,9 @@ class FalAIVideoConfig(BaseVideoConfig): error_value: Final[object] = response_data.get("error") error: Final[str | None] = error_value if isinstance(error_value, str) else None provider: Final[str] = custom_llm_provider or _FAL_AI_PROVIDER + model_path: Final[str | None] = _model_path_from_queue_url(response_data.get("response_url")) return VideoObject( - id=encode_video_id_with_provider(_response_string(response_data, "request_id"), provider), + id=encode_video_id_with_provider(_response_string(response_data, "request_id"), provider, model_path), object="video", status="failed" if error else status, created_at=0, diff --git a/tests/test_litellm/llms/fal_ai/videos/test_fal_ai_video_transformation.py b/tests/test_litellm/llms/fal_ai/videos/test_fal_ai_video_transformation.py index d4b058ddcf6..f367fa331d5 100644 --- a/tests/test_litellm/llms/fal_ai/videos/test_fal_ai_video_transformation.py +++ b/tests/test_litellm/llms/fal_ai/videos/test_fal_ai_video_transformation.py @@ -51,6 +51,14 @@ class TestFalAIVideoTransformation: "aspect_ratio": "1:1", } assert self.config.map_openai_params({"size": "720p"}, MODEL, False) == {"resolution": "720p"} + assert self.config.map_openai_params({"size": "720x1280"}, MODEL, False) == { + "resolution": "720p", + "aspect_ratio": "9:16", + } + assert self.config.map_openai_params({"size": "1080x1920"}, MODEL, False) == { + "resolution": "1080p", + "aspect_ratio": "9:16", + } def test_map_openai_params_rejects_non_url_input_reference(self): with pytest.raises(ValueError, match="public image URL"): @@ -165,6 +173,35 @@ class TestFalAIVideoTransformation: assert video.status == expected_status assert video.created_at == 0 + def test_status_response_id_stays_pollable(self): + response = Mock(spec=httpx.Response) + response.json.return_value = { + "request_id": "abc", + "status": "IN_PROGRESS", + "response_url": "https://queue.fal.run/bytedance/seedance-2.5/requests/abc", + } + + video = self.config.transform_video_status_retrieve_response( + raw_response=response, + logging_obj=self.logging_obj, + custom_llm_provider="fal_ai", + ) + + status_url, _ = self.config.transform_video_status_retrieve_request( + video_id=video.id, + api_base="https://queue.fal.run", + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + content_url, _ = self.config.transform_video_content_request( + video_id=video.id, + api_base="https://queue.fal.run", + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + assert status_url == "https://queue.fal.run/bytedance/seedance-2.5/requests/abc/status" + assert content_url == "https://queue.fal.run/bytedance/seedance-2.5/requests/abc" + def test_status_response_error(self): response = Mock(spec=httpx.Response) response.json.return_value = { From c359ef763eae44523cace811bcb1999992250bde Mon Sep 17 00:00:00 2001 From: kerry Date: Sat, 19 Sep 2026 17:01:09 +0000 Subject: [PATCH 4/9] fix(fal_ai): keep model in polled video ids and pick resolution from the short side Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/llms/fal_ai/videos/transformation.py | 28 ++++++-- .../test_fal_ai_video_transformation.py | 64 +++++++++++-------- 2 files changed, 58 insertions(+), 34 deletions(-) diff --git a/litellm/llms/fal_ai/videos/transformation.py b/litellm/llms/fal_ai/videos/transformation.py index 4ca1f918c90..766316a5b18 100644 --- a/litellm/llms/fal_ai/videos/transformation.py +++ b/litellm/llms/fal_ai/videos/transformation.py @@ -79,12 +79,22 @@ def _resolution_for_short_side(short_side: int) -> str: return next((resolution for threshold, resolution in _RESOLUTION_TIERS if short_side <= threshold), "4k") -def _model_path_from_queue_url(url: object) -> str | None: - if not isinstance(url, str) or not url: +def _model_path_from_request_url(raw_response: httpx.Response) -> str | None: + segments: Final[tuple[str, ...]] = tuple(segment for segment in raw_response.request.url.path.split("/") if segment) + if "requests" not in segments: return None - path: Final[str] = httpx.URL(url).path.strip("/") - model_path, separator, _ = path.partition("/requests/") - return model_path if separator and model_path else None + model_segments: Final[tuple[str, ...]] = segments[: segments.index("requests")] + segment_count: Final[int] = 3 if len(model_segments) >= 3 and model_segments[-3] in _QUEUE_NAMESPACES else 2 + return "/".join(model_segments[-segment_count:]) if len(model_segments) >= segment_count else None + + +def _request_id_from_request_url(raw_response: httpx.Response) -> str | None: + segments: Final[tuple[str, ...]] = tuple(segment for segment in raw_response.request.url.path.split("/") if segment) + if "requests" not in segments: + return None + request_index: Final[int] = segments.index("requests") + request_id_index: Final[int] = request_index + 1 + return segments[request_id_index] if len(segments) > request_id_index else None def _size_params(size: object) -> Mapping[str, str]: @@ -295,12 +305,16 @@ class FalAIVideoConfig(BaseVideoConfig): error_value: Final[object] = response_data.get("error") error: Final[str | None] = error_value if isinstance(error_value, str) else None provider: Final[str] = custom_llm_provider or _FAL_AI_PROVIDER - model_path: Final[str | None] = _model_path_from_queue_url(response_data.get("response_url")) + model_path: Final[str | None] = _model_path_from_request_url(raw_response) + request_id: Final[str] = _response_string(response_data, "request_id") or ( + _request_id_from_request_url(raw_response) or "" + ) return VideoObject( - id=encode_video_id_with_provider(_response_string(response_data, "request_id"), provider, model_path), + id=encode_video_id_with_provider(request_id, provider, model_path), object="video", status="failed" if error else status, created_at=0, + model=model_path, error=( {"code": "fal_error", "message": error} if error else None # mutable-ok: VideoObject requires a dict ), diff --git a/tests/test_litellm/llms/fal_ai/videos/test_fal_ai_video_transformation.py b/tests/test_litellm/llms/fal_ai/videos/test_fal_ai_video_transformation.py index f367fa331d5..8e0c68e30bb 100644 --- a/tests/test_litellm/llms/fal_ai/videos/test_fal_ai_video_transformation.py +++ b/tests/test_litellm/llms/fal_ai/videos/test_fal_ai_video_transformation.py @@ -161,8 +161,8 @@ class TestFalAIVideoTransformation: ], ) def test_status_response_mapping(self, response_data, expected_status): - response = Mock(spec=httpx.Response) - response.json.return_value = response_data + status_url = "https://queue.fal.run/bytedance/seedance-2.5/requests/abc/status" + response = httpx.Response(200, json=response_data, request=httpx.Request("GET", status_url)) video = self.config.transform_video_status_retrieve_response( raw_response=response, @@ -172,43 +172,32 @@ class TestFalAIVideoTransformation: assert video.status == expected_status assert video.created_at == 0 + decoded = decode_video_id_with_provider(video.id) + assert decoded["model_id"] == "bytedance/seedance-2.5" + assert decoded["video_id"] == "abc" - def test_status_response_id_stays_pollable(self): - response = Mock(spec=httpx.Response) - response.json.return_value = { - "request_id": "abc", - "status": "IN_PROGRESS", - "response_url": "https://queue.fal.run/bytedance/seedance-2.5/requests/abc", - } - - video = self.config.transform_video_status_retrieve_response( - raw_response=response, - logging_obj=self.logging_obj, - custom_llm_provider="fal_ai", - ) - - status_url, _ = self.config.transform_video_status_retrieve_request( + poll_url, _ = self.config.transform_video_status_retrieve_request( video_id=video.id, api_base="https://queue.fal.run", litellm_params=GenericLiteLLMParams(), headers={}, ) - content_url, _ = self.config.transform_video_content_request( - video_id=video.id, - api_base="https://queue.fal.run", - litellm_params=GenericLiteLLMParams(), - headers={}, - ) - assert status_url == "https://queue.fal.run/bytedance/seedance-2.5/requests/abc/status" - assert content_url == "https://queue.fal.run/bytedance/seedance-2.5/requests/abc" + assert poll_url == status_url def test_status_response_error(self): - response = Mock(spec=httpx.Response) - response.json.return_value = { + response_data = { "request_id": "abc", "status": "COMPLETED", "error": "generation failed", } + response = httpx.Response( + 200, + json=response_data, + request=httpx.Request( + "GET", + "https://queue.fal.run/bytedance/seedance-2.5/requests/abc/status", + ), + ) video = self.config.transform_video_status_retrieve_response( raw_response=response, @@ -219,6 +208,27 @@ class TestFalAIVideoTransformation: assert video.status == "failed" assert video.error == {"code": "fal_error", "message": "generation failed"} + def test_status_response_uses_namespaced_request_url(self): + response = httpx.Response( + 200, + json={"status": "IN_PROGRESS"}, + request=httpx.Request( + "GET", + "https://example.com/proxy/workflows/owner/app/requests/xyz/status", + ), + ) + + video = self.config.transform_video_status_retrieve_response( + raw_response=response, + logging_obj=self.logging_obj, + custom_llm_provider="fal_ai", + ) + + decoded = decode_video_id_with_provider(video.id) + assert decoded["model_id"] == "workflows/owner/app" + assert decoded["video_id"] == "xyz" + assert video.model == "workflows/owner/app" + def test_content_response_downloads_video_url(self, monkeypatch): content_response = httpx.Response( 200, From aa5f0858f75b3e074264e0266f87b70a6cb70391 Mon Sep 17 00:00:00 2001 From: kerry Date: Sat, 19 Sep 2026 17:13:41 +0000 Subject: [PATCH 5/9] test(pricing): allow video endpoint and rates Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- tests/test_litellm/test_utils.py | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py index b40c10de428..d694c08510a 100644 --- a/tests/test_litellm/test_utils.py +++ b/tests/test_litellm/test_utils.py @@ -940,6 +940,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "/v1/audio/transcriptions", "/v1/audio/speech", "/v1/ocr", + "/v1/videos", "/vertex_ai/live", "/v1/listen", "/v1beta/interactions", @@ -1069,6 +1070,9 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): # Add any model IDs that should be exempt from the cost validation # Example: "expensive-model-id", "runwayml/seedance2", # 4K output is 150 credits/second = $1.50/second + "fal_ai/bytedance/seedance-2.0/text-to-video", + "fal_ai/bytedance/seedance-2.0/image-to-video", + "fal_ai/bytedance/seedance-2.0/reference-to-video", ] is_valid, violations = validate_model_cost_values(actual_json, exceptions) From e0b455e94e83dfedad0364aedc6b4cdfcb59acb0 Mon Sep 17 00:00:00 2001 From: kerry Date: Sat, 19 Sep 2026 17:27:44 +0000 Subject: [PATCH 6/9] fix(fal_ai): read only the documented FAL_AI_API_KEY env var Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/llms/fal_ai/videos/transformation.py | 5 ++--- .../test_fal_ai_video_transformation.py | 20 +++++++++++++++++++ 2 files changed, 22 insertions(+), 3 deletions(-) diff --git a/litellm/llms/fal_ai/videos/transformation.py b/litellm/llms/fal_ai/videos/transformation.py index 766316a5b18..98528c82f6b 100644 --- a/litellm/llms/fal_ai/videos/transformation.py +++ b/litellm/llms/fal_ai/videos/transformation.py @@ -207,10 +207,9 @@ class FalAIVideoConfig(BaseVideoConfig): api_key or (litellm_params.api_key if litellm_params is not None else None) or get_secret_str("FAL_AI_API_KEY") - or get_secret_str("FAL_KEY") ) if not final_api_key: - raise ValueError("fal.ai API key is required") + raise ValueError("FAL_AI_API_KEY is not set") validated_headers: Final[_VideoHeaders] = { **headers, "Authorization": f"Key {final_api_key}", @@ -224,7 +223,7 @@ class FalAIVideoConfig(BaseVideoConfig): api_base: str | None, litellm_params: _VideoParams, ) -> str: - return (api_base or get_secret_str("FAL_AI_QUEUE_API_BASE") or "https://queue.fal.run").rstrip("/") + return (api_base or "https://queue.fal.run").rstrip("/") def transform_video_create_request( self, diff --git a/tests/test_litellm/llms/fal_ai/videos/test_fal_ai_video_transformation.py b/tests/test_litellm/llms/fal_ai/videos/test_fal_ai_video_transformation.py index 8e0c68e30bb..5e2e4532265 100644 --- a/tests/test_litellm/llms/fal_ai/videos/test_fal_ai_video_transformation.py +++ b/tests/test_litellm/llms/fal_ai/videos/test_fal_ai_video_transformation.py @@ -91,6 +91,26 @@ class TestFalAIVideoTransformation: } assert "model" not in body + def test_get_complete_url_respects_api_base_override(self): + url = self.config.get_complete_url( + model=MODEL, + api_base="https://proxy.internal/", + litellm_params={}, + ) + + assert url == "https://proxy.internal" + + def test_validate_environment_requires_fal_ai_api_key(self, monkeypatch): + monkeypatch.setattr(fal_video_module, "get_secret_str", lambda _: None) + + with pytest.raises(ValueError, match="FAL_AI_API_KEY is not set"): + self.config.validate_environment( + headers={}, + model=MODEL, + api_key=None, + litellm_params=GenericLiteLLMParams(), + ) + def test_transform_video_create_response_encodes_model_and_usage(self): response = Mock(spec=httpx.Response) response.json.return_value = {"request_id": "abc"} From 85a6a8e2063ef0932dc40b19c97eb0120a0c4235 Mon Sep 17 00:00:00 2001 From: kerry Date: Sat, 19 Sep 2026 18:22:58 +0000 Subject: [PATCH 7/9] test(e2e): cover fal Seedance video create, poll and download Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../llm_nonconversational.yaml | 1 + tests/e2e/coverage_registry/schema.py | 2 + .../LLM_TRANSLATION_COVERAGE_MATRIX.md | 2 + tests/e2e/llm_translation/endpoints_client.py | 36 +++++++++- .../test_video_generation_e2e.py | 69 +++++++++++++++++++ 5 files changed, 109 insertions(+), 1 deletion(-) create mode 100644 tests/e2e/llm_translation/test_video_generation_e2e.py diff --git a/tests/e2e/coverage_registry/llm_nonconversational.yaml b/tests/e2e/coverage_registry/llm_nonconversational.yaml index 50f9b9808b2..6970567b6f0 100644 --- a/tests/e2e/coverage_registry/llm_nonconversational.yaml +++ b/tests/e2e/coverage_registry/llm_nonconversational.yaml @@ -80,6 +80,7 @@ - {id: llm.images_generations.vertex.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: images_generations, route: vertex, capability: basic, streaming: nonstream, assertions: [works], source: "vertex_ai/image_generation/image_generation_handler.py", rationale: "Vertex Imagen"} - {id: llm.images_generations.bedrock.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: images_generations, route: bedrock_converse, capability: basic, streaming: nonstream, assertions: [works], source: "bedrock/image_generation/image_handler.py", rationale: "Bedrock Titan Image"} - {id: llm.images_generations.black_forest_labs.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: images_generations, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "black_forest_labs/image_generation/handler.py", rationale: "BFL Flux via OpenAI-compat"} +- {id: llm.videos.fal_ai.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: videos, route: fal_ai, capability: basic, streaming: nonstream, assertions: [works], source: "test_video_generation_e2e.py", rationale: "fal queue video create, poll, content download"} - {id: llm.audio_speech.openai.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: audio_speech, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "test_audio_speech_e2e.py:22", rationale: "OpenAI TTS binary audio"} - {id: llm.audio_speech.openai.basic.stream.works, module: llm, tier: P1, subject_endpoint: audio_speech, route: openai, capability: basic, streaming: stream, assertions: [works], source: "proxy_server.py:9043", rationale: "TTS streaming chunk generator"} - {id: llm.audio_speech.openai.input_validation.nonstream.works, module: llm, tier: P1, subject_endpoint: audio_speech, route: openai, capability: input_validation, streaming: nonstream, assertions: [works], source: "vendor strategy §9.6 / LIT-4778", rationale: "TTS missing input/model, invalid voice, empty input rejected"} diff --git a/tests/e2e/coverage_registry/schema.py b/tests/e2e/coverage_registry/schema.py index fa6dad90126..3ae17432863 100644 --- a/tests/e2e/coverage_registry/schema.py +++ b/tests/e2e/coverage_registry/schema.py @@ -44,6 +44,7 @@ LlmEndpoint = Literal[ "vector_stores", "ocr", "bedrock_native", + "videos", ] LlmRoute = Literal[ @@ -53,6 +54,7 @@ LlmRoute = Literal[ "bedrock_converse", "bedrock_invoke", "cohere", + "fal_ai", "gemini", "hosted_vllm", "openai", diff --git a/tests/e2e/llm_translation/LLM_TRANSLATION_COVERAGE_MATRIX.md b/tests/e2e/llm_translation/LLM_TRANSLATION_COVERAGE_MATRIX.md index 44d6e79122e..178af054f2b 100644 --- a/tests/e2e/llm_translation/LLM_TRANSLATION_COVERAGE_MATRIX.md +++ b/tests/e2e/llm_translation/LLM_TRANSLATION_COVERAGE_MATRIX.md @@ -48,6 +48,7 @@ most likely to silently break and the one a mock can't prove works. |----------|---------------|-----------|------------|-------------|--------| | Chat | live (spend suite) | live (spend suite) | gap | live | partial | | Embeddings | live (spend suite) | n/a | n/a | live | covered | +| Video | live (fal.ai Seedance) | n/a | n/a | - | partial | | Responses / image / audio / rerank / realtime | - | - | - | - | gap | ## This suite's files @@ -61,6 +62,7 @@ most likely to silently break and the one a mock can't prove works. | `test_anthropic_passthrough_streaming_logs_cost` | anthropic native, stream, cost | | `test_anthropic_passthrough_tool_call_logs_cost` | anthropic native, tool call, cost | | `test_vertex_passthrough_via_managed_model_logs_cost` | vertex_ai native, non-stream, cost | +| `test_fal_seedance_video_completes_and_downloads` | fal.ai Seedance video create, poll, and content download | Vertex keeps the credential on the proxy like gemini/anthropic, but the deployment is added at runtime instead of declared in the gateway config: the test POSTs `/model/new` diff --git a/tests/e2e/llm_translation/endpoints_client.py b/tests/e2e/llm_translation/endpoints_client.py index 4d2c73e7078..165a83e76c0 100644 --- a/tests/e2e/llm_translation/endpoints_client.py +++ b/tests/e2e/llm_translation/endpoints_client.py @@ -13,7 +13,7 @@ from dataclasses import dataclass from typing import Literal from e2e_config import SLOW_PROVIDER_TIMEOUT_SECONDS -from e2e_http import BinaryStream, Result, StreamingResponse +from e2e_http import BinaryStream, NoBody, Result, StreamingResponse from models import CacheControl, ChatMessage, LiteLLMParamsBody, RichMessage, TextBlock from proxy_client import ProxyClient from pydantic import BaseModel @@ -26,6 +26,8 @@ __all__ = [ "TextBlock", "TranscriptionForm", "TranscriptionResult", + "VideoObject", + "VideoRequest", ] @@ -127,6 +129,13 @@ class ImageRequest(BaseModel): size: str = "1024x1024" +class VideoRequest(BaseModel): + model: str + prompt: str + seconds: str = "4" + size: str = "1280x720" + + class ImageEditForm(BaseModel): model: str prompt: str @@ -267,6 +276,12 @@ class ImagesResult(BaseModel): data: list[ImageItem] = [] +class VideoObject(BaseModel): + id: str + status: str + model: str | None = None + + class TranscriptionResult(BaseModel): text: str = "" @@ -440,6 +455,25 @@ class EndpointsClient: "/v1/images/generations", key, ImageRequest(model=model, prompt=prompt) ) + def videos(self, key: str, model: str, prompt: str) -> StreamingResponse: + return self._send( + "/v1/videos", key, VideoRequest(model=model, prompt=prompt) + ) + + def video_status(self, key: str, video_id: str) -> Result[VideoObject]: + return self.proxy.transport.get( + f"/v1/videos/{video_id}", + headers=self.proxy.transport.bearer(key), + params=NoBody(), + response_type=VideoObject, + ) + + def video_content(self, key: str, video_id: str) -> StreamingResponse: + return self.proxy.transport.download( + f"/v1/videos/{video_id}/content", + headers=self.proxy.transport.bearer(key), + ) + def image_edit( self, key: str, model: str, prompt: str, image: bytes, *, filename: str = "image.png" ) -> Result[ImagesResult]: diff --git a/tests/e2e/llm_translation/test_video_generation_e2e.py b/tests/e2e/llm_translation/test_video_generation_e2e.py new file mode 100644 index 00000000000..b65529aa260 --- /dev/null +++ b/tests/e2e/llm_translation/test_video_generation_e2e.py @@ -0,0 +1,69 @@ +"""Live e2e: POST /v1/videos creates a video and serves its content. + +Registers a fal.ai Seedance deployment at runtime, polls the queued video until it +completes, and asserts the generated content is returned as binary data. +""" + +from __future__ import annotations + +import time +from typing import Final + +import pytest +from e2e_config import unique_marker +from e2e_http import require_successful_call, unwrap +from endpoints_client import EndpointsClient, VideoObject +from lifecycle import ResourceManager +from models import LiteLLMParamsBody + +pytestmark = pytest.mark.e2e + +_POLL_INTERVAL_SECONDS: Final[float] = 5.0 +_POLL_TIMEOUT_SECONDS: Final[float] = 600.0 + + +def _wait_for_completion( + endpoints_client: EndpointsClient, key: str, created: VideoObject +) -> VideoObject: + deadline = time.monotonic() + _POLL_TIMEOUT_SECONDS + while time.monotonic() < deadline: + status = unwrap(endpoints_client.video_status(key, created.id)) + assert status.id == created.id + if status.status == "completed": + return status + if status.status == "failed": + pytest.fail(f"fal.ai video generation failed: {status}") + time.sleep(_POLL_INTERVAL_SECONDS) + pytest.fail(f"fal.ai video {created.id!r} did not complete within {_POLL_TIMEOUT_SECONDS}s") + + +class TestVideoGeneration: + @pytest.mark.covers("llm.videos.fal_ai.basic.nonstream.works") + def test_fal_seedance_video_completes_and_downloads( + self, endpoints_client: EndpointsClient, resources: ResourceManager + ) -> None: + model = f"e2e-fal-video-{unique_marker()}" + model_id = endpoints_client.create_model( + model, + LiteLLMParamsBody( + model="fal_ai/bytedance/seedance-2.5/text-to-video", + api_key="os.environ/FAL_AI_API_KEY", + ), + ) + resources.defer(lambda: endpoints_client.delete_model(model_id)) + key = resources.key() + + result = endpoints_client.videos( + key, model, "a red fox running through snow at dawn" + ) + require_successful_call(result) + created = VideoObject.model_validate_json(result.body) + assert created.id + assert created.model + + _wait_for_completion(endpoints_client, key, created) + + content = endpoints_client.video_content(key, created.id) + require_successful_call(content) + assert len(content.body) > 0 + assert not (content.content_type or "").startswith("application/json") From 9a63e06c63e60b3b30126aa7461115a2ccc041fe Mon Sep 17 00:00:00 2001 From: kerry Date: Sat, 19 Sep 2026 23:28:51 +0000 Subject: [PATCH 8/9] test(integration): cover fal Seedance video queue wire contract Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../llm_nonconversational.yaml | 1 - tests/e2e/coverage_registry/schema.py | 2 - .../LLM_TRANSLATION_COVERAGE_MATRIX.md | 2 - tests/e2e/llm_translation/endpoints_client.py | 36 +--------- .../test_video_generation_e2e.py | 69 ------------------ tests/integration/contracts.json | 3 + .../providers/test_fal_ai_video_wire.py | 72 +++++++++++++++++++ 7 files changed, 76 insertions(+), 109 deletions(-) delete mode 100644 tests/e2e/llm_translation/test_video_generation_e2e.py create mode 100644 tests/integration/providers/test_fal_ai_video_wire.py diff --git a/tests/e2e/coverage_registry/llm_nonconversational.yaml b/tests/e2e/coverage_registry/llm_nonconversational.yaml index 6970567b6f0..50f9b9808b2 100644 --- a/tests/e2e/coverage_registry/llm_nonconversational.yaml +++ b/tests/e2e/coverage_registry/llm_nonconversational.yaml @@ -80,7 +80,6 @@ - {id: llm.images_generations.vertex.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: images_generations, route: vertex, capability: basic, streaming: nonstream, assertions: [works], source: "vertex_ai/image_generation/image_generation_handler.py", rationale: "Vertex Imagen"} - {id: llm.images_generations.bedrock.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: images_generations, route: bedrock_converse, capability: basic, streaming: nonstream, assertions: [works], source: "bedrock/image_generation/image_handler.py", rationale: "Bedrock Titan Image"} - {id: llm.images_generations.black_forest_labs.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: images_generations, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "black_forest_labs/image_generation/handler.py", rationale: "BFL Flux via OpenAI-compat"} -- {id: llm.videos.fal_ai.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: videos, route: fal_ai, capability: basic, streaming: nonstream, assertions: [works], source: "test_video_generation_e2e.py", rationale: "fal queue video create, poll, content download"} - {id: llm.audio_speech.openai.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: audio_speech, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "test_audio_speech_e2e.py:22", rationale: "OpenAI TTS binary audio"} - {id: llm.audio_speech.openai.basic.stream.works, module: llm, tier: P1, subject_endpoint: audio_speech, route: openai, capability: basic, streaming: stream, assertions: [works], source: "proxy_server.py:9043", rationale: "TTS streaming chunk generator"} - {id: llm.audio_speech.openai.input_validation.nonstream.works, module: llm, tier: P1, subject_endpoint: audio_speech, route: openai, capability: input_validation, streaming: nonstream, assertions: [works], source: "vendor strategy §9.6 / LIT-4778", rationale: "TTS missing input/model, invalid voice, empty input rejected"} diff --git a/tests/e2e/coverage_registry/schema.py b/tests/e2e/coverage_registry/schema.py index 3ae17432863..fa6dad90126 100644 --- a/tests/e2e/coverage_registry/schema.py +++ b/tests/e2e/coverage_registry/schema.py @@ -44,7 +44,6 @@ LlmEndpoint = Literal[ "vector_stores", "ocr", "bedrock_native", - "videos", ] LlmRoute = Literal[ @@ -54,7 +53,6 @@ LlmRoute = Literal[ "bedrock_converse", "bedrock_invoke", "cohere", - "fal_ai", "gemini", "hosted_vllm", "openai", diff --git a/tests/e2e/llm_translation/LLM_TRANSLATION_COVERAGE_MATRIX.md b/tests/e2e/llm_translation/LLM_TRANSLATION_COVERAGE_MATRIX.md index 178af054f2b..44d6e79122e 100644 --- a/tests/e2e/llm_translation/LLM_TRANSLATION_COVERAGE_MATRIX.md +++ b/tests/e2e/llm_translation/LLM_TRANSLATION_COVERAGE_MATRIX.md @@ -48,7 +48,6 @@ most likely to silently break and the one a mock can't prove works. |----------|---------------|-----------|------------|-------------|--------| | Chat | live (spend suite) | live (spend suite) | gap | live | partial | | Embeddings | live (spend suite) | n/a | n/a | live | covered | -| Video | live (fal.ai Seedance) | n/a | n/a | - | partial | | Responses / image / audio / rerank / realtime | - | - | - | - | gap | ## This suite's files @@ -62,7 +61,6 @@ most likely to silently break and the one a mock can't prove works. | `test_anthropic_passthrough_streaming_logs_cost` | anthropic native, stream, cost | | `test_anthropic_passthrough_tool_call_logs_cost` | anthropic native, tool call, cost | | `test_vertex_passthrough_via_managed_model_logs_cost` | vertex_ai native, non-stream, cost | -| `test_fal_seedance_video_completes_and_downloads` | fal.ai Seedance video create, poll, and content download | Vertex keeps the credential on the proxy like gemini/anthropic, but the deployment is added at runtime instead of declared in the gateway config: the test POSTs `/model/new` diff --git a/tests/e2e/llm_translation/endpoints_client.py b/tests/e2e/llm_translation/endpoints_client.py index 165a83e76c0..4d2c73e7078 100644 --- a/tests/e2e/llm_translation/endpoints_client.py +++ b/tests/e2e/llm_translation/endpoints_client.py @@ -13,7 +13,7 @@ from dataclasses import dataclass from typing import Literal from e2e_config import SLOW_PROVIDER_TIMEOUT_SECONDS -from e2e_http import BinaryStream, NoBody, Result, StreamingResponse +from e2e_http import BinaryStream, Result, StreamingResponse from models import CacheControl, ChatMessage, LiteLLMParamsBody, RichMessage, TextBlock from proxy_client import ProxyClient from pydantic import BaseModel @@ -26,8 +26,6 @@ __all__ = [ "TextBlock", "TranscriptionForm", "TranscriptionResult", - "VideoObject", - "VideoRequest", ] @@ -129,13 +127,6 @@ class ImageRequest(BaseModel): size: str = "1024x1024" -class VideoRequest(BaseModel): - model: str - prompt: str - seconds: str = "4" - size: str = "1280x720" - - class ImageEditForm(BaseModel): model: str prompt: str @@ -276,12 +267,6 @@ class ImagesResult(BaseModel): data: list[ImageItem] = [] -class VideoObject(BaseModel): - id: str - status: str - model: str | None = None - - class TranscriptionResult(BaseModel): text: str = "" @@ -455,25 +440,6 @@ class EndpointsClient: "/v1/images/generations", key, ImageRequest(model=model, prompt=prompt) ) - def videos(self, key: str, model: str, prompt: str) -> StreamingResponse: - return self._send( - "/v1/videos", key, VideoRequest(model=model, prompt=prompt) - ) - - def video_status(self, key: str, video_id: str) -> Result[VideoObject]: - return self.proxy.transport.get( - f"/v1/videos/{video_id}", - headers=self.proxy.transport.bearer(key), - params=NoBody(), - response_type=VideoObject, - ) - - def video_content(self, key: str, video_id: str) -> StreamingResponse: - return self.proxy.transport.download( - f"/v1/videos/{video_id}/content", - headers=self.proxy.transport.bearer(key), - ) - def image_edit( self, key: str, model: str, prompt: str, image: bytes, *, filename: str = "image.png" ) -> Result[ImagesResult]: diff --git a/tests/e2e/llm_translation/test_video_generation_e2e.py b/tests/e2e/llm_translation/test_video_generation_e2e.py deleted file mode 100644 index b65529aa260..00000000000 --- a/tests/e2e/llm_translation/test_video_generation_e2e.py +++ /dev/null @@ -1,69 +0,0 @@ -"""Live e2e: POST /v1/videos creates a video and serves its content. - -Registers a fal.ai Seedance deployment at runtime, polls the queued video until it -completes, and asserts the generated content is returned as binary data. -""" - -from __future__ import annotations - -import time -from typing import Final - -import pytest -from e2e_config import unique_marker -from e2e_http import require_successful_call, unwrap -from endpoints_client import EndpointsClient, VideoObject -from lifecycle import ResourceManager -from models import LiteLLMParamsBody - -pytestmark = pytest.mark.e2e - -_POLL_INTERVAL_SECONDS: Final[float] = 5.0 -_POLL_TIMEOUT_SECONDS: Final[float] = 600.0 - - -def _wait_for_completion( - endpoints_client: EndpointsClient, key: str, created: VideoObject -) -> VideoObject: - deadline = time.monotonic() + _POLL_TIMEOUT_SECONDS - while time.monotonic() < deadline: - status = unwrap(endpoints_client.video_status(key, created.id)) - assert status.id == created.id - if status.status == "completed": - return status - if status.status == "failed": - pytest.fail(f"fal.ai video generation failed: {status}") - time.sleep(_POLL_INTERVAL_SECONDS) - pytest.fail(f"fal.ai video {created.id!r} did not complete within {_POLL_TIMEOUT_SECONDS}s") - - -class TestVideoGeneration: - @pytest.mark.covers("llm.videos.fal_ai.basic.nonstream.works") - def test_fal_seedance_video_completes_and_downloads( - self, endpoints_client: EndpointsClient, resources: ResourceManager - ) -> None: - model = f"e2e-fal-video-{unique_marker()}" - model_id = endpoints_client.create_model( - model, - LiteLLMParamsBody( - model="fal_ai/bytedance/seedance-2.5/text-to-video", - api_key="os.environ/FAL_AI_API_KEY", - ), - ) - resources.defer(lambda: endpoints_client.delete_model(model_id)) - key = resources.key() - - result = endpoints_client.videos( - key, model, "a red fox running through snow at dawn" - ) - require_successful_call(result) - created = VideoObject.model_validate_json(result.body) - assert created.id - assert created.model - - _wait_for_completion(endpoints_client, key, created) - - content = endpoints_client.video_content(key, created.id) - require_successful_call(content) - assert len(content.body) > 0 - assert not (content.content_type or "").startswith("application/json") diff --git a/tests/integration/contracts.json b/tests/integration/contracts.json index 6958ade50f7..01f7af6e8fe 100644 --- a/tests/integration/contracts.json +++ b/tests/integration/contracts.json @@ -160,6 +160,9 @@ "other.provider_wire.anthropic.tool_history_system_cache_and_internal_fields", "quota_management.spend_tracking.cache_tokens.disjoint_classes_use_explicit_rates" ], + "tests/integration/providers/test_fal_ai_video_wire.py::test_fal_video_create_status_and_content_follow_queue_wire_contract": [ + "other.provider_wire.fal_ai.video_queue_create_status_and_content_download" + ], "tests/integration/mcp/test_mcp_lifecycle.py::test_saved_headers_reach_real_mcp_tool_and_survive_unrelated_edit": [ "mcp.call_tool.saved_headers.reach_actual_transport" ], diff --git a/tests/integration/providers/test_fal_ai_video_wire.py b/tests/integration/providers/test_fal_ai_video_wire.py new file mode 100644 index 00000000000..c1a2655f0aa --- /dev/null +++ b/tests/integration/providers/test_fal_ai_video_wire.py @@ -0,0 +1,72 @@ +import json +import sys +import uuid +from typing import Final + +import pytest +from integration._support.client import Gateway +from integration._support.wire import Reply, Request, wire_server + +_MODEL: Final = "bytedance/seedance-2.5/text-to-video" +_MP4: Final = b"\x00\x00\x00\x18ftypmp42" + uuid.uuid4().bytes * 4 + + +@pytest.mark.covers("other.provider_wire.fal_ai.video_queue_create_status_and_content_download") +def test_fal_video_create_status_and_content_follow_queue_wire_contract(gateway: Gateway) -> None: + request_id: Final = "fal-req-" + uuid.uuid4().hex + + def respond(request: Request) -> Reply: + if request.target == f"/files/{request_id}.mp4": + assert request.method == "GET" + return Reply(body=_MP4, content_type="video/mp4") + assert request.headers["authorization"] == "Key synthetic-fal-key" + if request.method == "POST": + assert request.target == f"/{_MODEL}" + assert json.loads(request.body) == { + "prompt": "a cat playing volleyball on a beach", + "duration": "4", + "resolution": "720p", + "aspect_ratio": "16:9", + } + return Reply( + body=json.dumps({"status": "IN_QUEUE", "request_id": request_id, "queue_position": 0}).encode() + ) + assert request.method == "GET" + if request.target == f"/bytedance/seedance-2.5/requests/{request_id}/status": + return Reply(body=json.dumps({"status": "COMPLETED", "request_id": request_id}).encode()) + assert request.target == f"/bytedance/seedance-2.5/requests/{request_id}" + return Reply(body=json.dumps({"video": {"url": f"{wire_url}/files/{request_id}.mp4"}}).encode()) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + wire_url: Final = wire.url + model: Final = scenario.model( + model=f"fal_ai/{_MODEL}", + api_base=wire.url, + api_key="synthetic-fal-key", + ) + created: Final = gateway.post( + "/v1/videos", + { + "model": model, + "prompt": "a cat playing volleyball on a beach", + "seconds": "4", + "size": "1280x720", + }, + ) + assert created["status"] == "queued" + video_id: Final = created["id"] + assert isinstance(video_id, str) and video_id + status: Final = gateway.get(f"/v1/videos/{video_id}") + assert status["status"] == "completed" + status_id_matches_created_id: Final = status["id"] == video_id + sys.stdout.write(f"status_id_matches_created_id={status_id_matches_created_id}\n") + content: Final = gateway.request("GET", f"/v1/videos/{video_id}/content") + assert content.status_code == 200, content.text + assert content.headers["content-type"].startswith("video/mp4") + assert content.content == _MP4 + assert [(request.method, request.target) for request in wire.drain()] == [ + ("POST", f"/{_MODEL}"), + ("GET", f"/bytedance/seedance-2.5/requests/{request_id}/status"), + ("GET", f"/bytedance/seedance-2.5/requests/{request_id}"), + ("GET", f"/files/{request_id}.mp4"), + ] From 21a2ed62448ebda3ab9de1245b2550aa0bf164e0 Mon Sep 17 00:00:00 2001 From: kerry Date: Sat, 19 Sep 2026 23:29:25 +0000 Subject: [PATCH 9/9] test(integration): drop id diagnostic from fal video wire test Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- tests/integration/providers/test_fal_ai_video_wire.py | 3 --- 1 file changed, 3 deletions(-) diff --git a/tests/integration/providers/test_fal_ai_video_wire.py b/tests/integration/providers/test_fal_ai_video_wire.py index c1a2655f0aa..8c72810ffb6 100644 --- a/tests/integration/providers/test_fal_ai_video_wire.py +++ b/tests/integration/providers/test_fal_ai_video_wire.py @@ -1,5 +1,4 @@ import json -import sys import uuid from typing import Final @@ -58,8 +57,6 @@ def test_fal_video_create_status_and_content_follow_queue_wire_contract(gateway: assert isinstance(video_id, str) and video_id status: Final = gateway.get(f"/v1/videos/{video_id}") assert status["status"] == "completed" - status_id_matches_created_id: Final = status["id"] == video_id - sys.stdout.write(f"status_id_matches_created_id={status_id_matches_created_id}\n") content: Final = gateway.request("GET", f"/v1/videos/{video_id}/content") assert content.status_code == 200, content.text assert content.headers["content-type"].startswith("video/mp4")