Merge remote-tracking branch 'origin/litellm_internal_staging' into litellm_harden_retry_breadcrumb_credentials

This commit is contained in:
mateo-berri 2026-08-24 12:57:42 -07:00
commit a1f1aa9cb5
40 changed files with 1307 additions and 208 deletions

View file

@ -151,6 +151,7 @@ _VIDEO_CALL_TYPES: Final = frozenset(
}
)
_SPEECH_CALL_TYPES: Final = frozenset(
{
CallTypes.speech.value,
@ -1379,23 +1380,36 @@ def completion_cost(
if custom_pricing and litellm_logging_obj is not None:
_litellm_params = getattr(litellm_logging_obj, "litellm_params", None)
if _litellm_params is not None:
_metadata = _litellm_params.get("metadata", {}) or {}
_video_model_info = _metadata.get("model_info", None)
_video_model_info = next(
(
model_info
for _metadata_key in ("metadata", "litellm_metadata")
if (model_info := (_litellm_params.get(_metadata_key) or {}).get("model_info"))
is not None
),
None,
)
usage_obj = getattr(completion_response, "usage", None)
duration_seconds: float | None = None
video_resolution: str | None = None
provider_reported_cost: float | None = None
if completion_response is not None and usage_obj:
# Handle both dict and Pydantic Usage object
if isinstance(usage_obj, dict):
duration_seconds = usage_obj.get("duration_seconds", None)
_vr = usage_obj.get("video_resolution", None)
provider_reported_cost = usage_obj.get("provider_reported_cost_usd", None)
else:
duration_seconds = getattr(usage_obj, "duration_seconds", None)
_vr = getattr(usage_obj, "video_resolution", None)
provider_reported_cost = getattr(usage_obj, "provider_reported_cost_usd", None)
if _vr is not None:
video_resolution = str(_vr).strip().lower()
if _video_model_info is None and provider_reported_cost is not None:
return float(provider_reported_cost)
if duration_seconds is not None:
# Calculate cost based on video duration using video-specific cost calculation
from litellm.llms.openai.cost_calculation import (

View file

@ -2301,6 +2301,7 @@ def exception_type(
or custom_llm_provider == "custom_openai"
or custom_llm_provider in litellm.openai_compatible_providers
or custom_llm_provider == "mistral"
or custom_llm_provider == "runwayml"
):
_map_openai_exception(
model=model,

View file

@ -440,6 +440,16 @@ class AnthropicModelInfo(BaseLLMModelInfo):
"""
return AnthropicModelInfo._supports_model_capability(model, "thinking_always_on", custom_llm_provider)
@staticmethod
def _supports_legacy_thinking(model: str, custom_llm_provider: str) -> bool:
"""Whether ``model`` is an adaptive-thinking model that still accepts legacy
``thinking.type=enabled`` with ``budget_tokens`` (the Claude 4.6 family).
The model cost map is authoritative: an explicit ``supports_legacy_thinking``
entry resolved under ``custom_llm_provider``, or a ``fallback_generalizations``
rule for unmapped 4.6 ids. Absent flag means the model rejects the legacy shape.
"""
return AnthropicModelInfo._supports_model_capability(model, "supports_legacy_thinking", custom_llm_provider)
@staticmethod
def maybe_drop_disabled_thinking(
model: str,

View file

@ -379,13 +379,19 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
def _translate_legacy_thinking_for_adaptive_model(
model: str, optional_params: dict, custom_llm_provider: str
) -> None:
"""Translate legacy ``thinking.type=enabled`` to adaptive for 4.6/4.7.
Caller-provided ``output_config.effort`` is never overridden.
"""Translate legacy ``thinking.type=enabled`` to adaptive for the
adaptive-thinking models that reject it (4.7+ and the 5 families).
Models flagged ``supports_legacy_thinking`` (the 4.6 family) accept the
legacy shape natively, so it is forwarded verbatim and the caller's
``budget_tokens`` cap keeps applying. Caller-provided
``output_config.effort`` is never overridden.
"""
from litellm.llms.anthropic.chat.transformation import AnthropicConfig
if not AnthropicModelInfo._is_adaptive_thinking_model(model, custom_llm_provider):
return
if AnthropicModelInfo._supports_legacy_thinking(model, custom_llm_provider):
return
thinking: Final = optional_params.get("thinking")
if not isinstance(thinking, dict) or thinking.get("type") != "enabled":
return

View file

@ -134,14 +134,7 @@ def cost_per_second(model: str, custom_llm_provider: str | None, duration: float
def _video_resolution_to_cost_field_suffix(resolution: str) -> str | None:
"""
Map usage resolution to a safe suffix for ``output_cost_per_second_<suffix>`` keys.
Note: Currently only ``output_cost_per_second_1080p`` is explicitly declared in
ModelInfo (types/utils.py). Other resolution tiers (e.g., 720p, 4k) can be added
to model_prices_and_context_window.json but are not exposed via get_model_info()
until added to the ModelInfo TypedDict.
"""
"""Map usage resolution to a safe suffix for ``output_cost_per_second_<suffix>`` keys."""
r: Final = resolution.strip().lower()
if not r:
return None

View file

@ -1,5 +1,6 @@
from collections.abc import Mapping, Sequence
from datetime import datetime
from types import MappingProxyType
from typing import TYPE_CHECKING, Any, Final, Literal
import httpx
@ -33,6 +34,10 @@ else:
LiteLLMLoggingObj = Any
class RunwayMLError(BaseLLMException):
pass
class _RunwayTaskResponse(TypedDict, total=False):
id: ReadOnly[str]
status: ReadOnly[str]
@ -41,7 +46,8 @@ class _RunwayTaskResponse(TypedDict, total=False):
output: ReadOnly[Sequence[str] | str]
failureCode: ReadOnly[str]
failure: ReadOnly[str]
progress: ReadOnly[int]
progress: ReadOnly[float]
estimatedCost: ReadOnly[Mapping[str, float]]
class _VideoObjectData(TypedDict, extra_items=object):
@ -56,12 +62,54 @@ def _parse_runway_task_response(raw_response: httpx.Response) -> _RunwayTaskResp
return response_data
_USD_PER_CREDIT: Final = 0.01
_RESOLUTION_AREA_TIERS: Final[tuple[tuple[int, str], ...]] = (
(600_000, "480p"),
(1_500_000, "720p"),
(4_000_000, "1080p"),
)
def _ratio_to_resolution(ratio: object) -> str | None:
if not isinstance(ratio, str) or ":" not in ratio:
return None
width_str, _, height_str = ratio.partition(":")
if not (width_str.isdigit() and height_str.isdigit()):
return None
area: Final = int(width_str) * int(height_str)
return next((label for threshold, label in _RESOLUTION_AREA_TIERS if area < threshold), "4k")
def _duration_seconds(seconds: str | None) -> float | None:
if not seconds:
return None
try:
return float(seconds)
except ValueError:
return None
def _estimated_cost_usd(response_data: _RunwayTaskResponse) -> float | None:
estimated_cost: Final = response_data.get("estimatedCost")
if not isinstance(estimated_cost, Mapping):
return None
credits: Final = estimated_cost.get("credits")
if not isinstance(credits, (int, float)):
return None
return float(credits) * _USD_PER_CREDIT
def _progress_percent(progress: float) -> int:
return min(100, max(0, round(float(progress) * 100)))
class RunwayMLVideoConfig(BaseVideoConfig):
"""
Configuration class for RunwayML video generation.
RunwayML uses a task-based API where:
1. POST /v1/image_to_video creates a task
1. POST /v1/text_to_video, /v1/image_to_video, or /v1/video_to_video creates a task
2. The task returns immediately with a task ID
3. Client must poll or wait for task completion
"""
@ -195,31 +243,36 @@ class RunwayMLVideoConfig(BaseVideoConfig):
"""
Transform the video creation request for RunwayML API.
RunwayML expects:
{
"model": "gen4_turbo",
"promptImage": "https://... or data:image/...",
"promptText": "description",
"ratio": "1280:720",
"duration": 5
}
RunwayML has three generation endpoints discriminated by which input is
present, and each request body rejects unknown fields:
- /text_to_video: promptText only (rejects promptImage)
- /image_to_video: promptImage (+ optional promptText)
- /video_to_video: promptVideo or videoUri (rejects promptImage)
"""
# Build the request data
merged_params: Final = MappingProxyType(
{
"model": model,
"promptText": prompt,
**video_create_optional_request_params,
}
)
endpoint: Final = self._select_generation_endpoint(merged_params)
request_data: Final[dict[str, object]] = {
"model": model,
"promptText": prompt,
key: value for key, value in merged_params.items() if endpoint == "image_to_video" or key != "promptImage"
}
# Add mapped parameters
request_data.update(video_create_optional_request_params)
# RunwayML uses JSON body, no files multipart
files_list: Final[RequestFiles] = []
# Append the specific endpoint for video generation
full_api_base: Final = f"{api_base}/image_to_video"
return request_data, files_list, f"{api_base}/{endpoint}"
return request_data, files_list, full_api_base
def _select_generation_endpoint(self, request_data: Mapping[str, object]) -> str:
if request_data.get("promptVideo") is not None or request_data.get("videoUri") is not None:
return "video_to_video"
if request_data.get("promptImage") is not None:
return "image_to_video"
return "text_to_video"
def transform_video_create_response(
self,
@ -285,13 +338,15 @@ class RunwayMLVideoConfig(BaseVideoConfig):
video_obj.id = encode_video_id_with_provider(video_obj.id, custom_llm_provider, model)
# Add usage data for cost tracking
usage_data: Final = {}
if video_obj and hasattr(video_obj, "seconds") and video_obj.seconds:
try:
usage_data["duration_seconds"] = float(video_obj.seconds)
except (ValueError, TypeError):
pass
video_obj.usage = usage_data
video_obj.usage = {
key: value
for key, value in (
("duration_seconds", _duration_seconds(video_obj.seconds)),
("video_resolution", _ratio_to_resolution(request_data.get("ratio") if request_data else None)),
("provider_reported_cost_usd", _estimated_cost_usd(response_data)),
)
if value is not None
}
return video_obj
@ -581,8 +636,9 @@ class RunwayMLVideoConfig(BaseVideoConfig):
if "completedAt" in response_data:
video_data["completed_at"] = self._parse_runway_timestamp(response_data.get("completedAt"))
if "progress" in response_data:
video_data["progress"] = response_data["progress"]
progress_value: Final = response_data.get("progress")
if progress_value is not None:
video_data["progress"] = _progress_percent(progress_value)
if "failureCode" in response_data or "failure" in response_data:
video_data["error"] = {
@ -646,9 +702,7 @@ class RunwayMLVideoConfig(BaseVideoConfig):
raise NotImplementedError("video extension is not supported for RunwayML")
def get_error_class(self, error_message: str, status_code: int, headers: dict | httpx.Headers) -> BaseLLMException:
from ...base_llm.chat.transformation import BaseLLMException
raise BaseLLMException(
return RunwayMLError(
status_code=status_code,
message=error_message,
headers=headers,

View file

@ -1019,6 +1019,7 @@
},
"anthropic.claude-opus-4-6-v1": {
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 6.25e-06,
"cache_creation_input_token_cost_above_1hr": 1e-05,
"cache_read_input_token_cost": 5e-07,
@ -1053,6 +1054,7 @@
},
"global.anthropic.claude-opus-4-6-v1": {
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 6.25e-06,
"cache_creation_input_token_cost_above_1hr": 1e-05,
"cache_read_input_token_cost": 5e-07,
@ -1087,6 +1089,7 @@
},
"us.anthropic.claude-opus-4-6-v1": {
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 6.875e-06,
"cache_creation_input_token_cost_above_1hr": 1.1e-05,
"cache_read_input_token_cost": 5.5e-07,
@ -1121,6 +1124,7 @@
},
"eu.anthropic.claude-opus-4-6-v1": {
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 6.875e-06,
"cache_creation_input_token_cost_above_1hr": 1.1e-05,
"cache_read_input_token_cost": 5.5e-07,
@ -1155,6 +1159,7 @@
},
"au.anthropic.claude-opus-4-6-v1": {
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 6.875e-06,
"cache_creation_input_token_cost_above_1hr": 1.1e-05,
"cache_read_input_token_cost": 5.5e-07,
@ -2233,6 +2238,7 @@
},
"anthropic.claude-sonnet-4-6": {
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 3.75e-06,
"cache_creation_input_token_cost_above_1hr": 6e-06,
"cache_read_input_token_cost": 3e-07,
@ -2266,6 +2272,7 @@
},
"global.anthropic.claude-sonnet-4-6": {
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 3.75e-06,
"cache_creation_input_token_cost_above_1hr": 6e-06,
"cache_read_input_token_cost": 3e-07,
@ -2299,6 +2306,7 @@
},
"us.anthropic.claude-sonnet-4-6": {
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 4.125e-06,
"cache_creation_input_token_cost_above_1hr": 6.6e-06,
"cache_read_input_token_cost": 3.3e-07,
@ -2332,6 +2340,7 @@
},
"eu.anthropic.claude-sonnet-4-6": {
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 4.125e-06,
"cache_creation_input_token_cost_above_1hr": 6.6e-06,
"cache_read_input_token_cost": 3.3e-07,
@ -2365,6 +2374,7 @@
},
"au.anthropic.claude-sonnet-4-6": {
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 4.125e-06,
"cache_creation_input_token_cost_above_1hr": 6.6e-06,
"cache_read_input_token_cost": 3.3e-07,
@ -2398,6 +2408,7 @@
},
"jp.anthropic.claude-sonnet-4-6": {
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 4.125e-06,
"cache_creation_input_token_cost_above_1hr": 6.6e-06,
"cache_read_input_token_cost": 3.3e-07,
@ -2950,6 +2961,7 @@
"azure_ai/claude-opus-4-6": {
"deprecation_date": "2027-02-02",
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"input_cost_per_token": 5e-06,
"output_cost_per_token": 2.5e-05,
"litellm_provider": "azure_ai",
@ -3181,6 +3193,7 @@
"azure_ai/claude-sonnet-4-6": {
"deprecation_date": "2027-02-10",
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 3.75e-06,
"cache_creation_input_token_cost_above_1hr": 6e-06,
"cache_read_input_token_cost": 3e-07,
@ -12489,6 +12502,7 @@
"search_context_size_medium": 0.01
},
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"supports_assistant_prefill": true,
"supports_computer_use": true,
"supports_function_calling": true,
@ -12698,6 +12712,7 @@
"search_context_size_medium": 0.01
},
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"supports_assistant_prefill": false,
"supports_computer_use": true,
"supports_function_calling": true,
@ -12735,6 +12750,7 @@
"search_context_size_medium": 0.01
},
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"supports_assistant_prefill": false,
"supports_computer_use": true,
"supports_function_calling": true,
@ -14724,6 +14740,7 @@
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_legacy_thinking": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_tool_choice": true
@ -14892,6 +14909,7 @@
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_legacy_thinking": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_tool_choice": true
@ -23427,6 +23445,7 @@
},
"github_copilot/claude-opus-4.6-fast": {
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"litellm_provider": "github_copilot",
"max_input_tokens": 128000,
"max_output_tokens": 16000,
@ -33810,6 +33829,7 @@
},
"openrouter/anthropic/claude-sonnet-4.6": {
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 3.75e-06,
"cache_creation_input_token_cost_above_200k_tokens": 7.5e-06,
"cache_read_input_token_cost": 3e-07,
@ -33854,6 +33874,7 @@
},
"openrouter/anthropic/claude-opus-4.6": {
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 6.25e-06,
"cache_read_input_token_cost": 5e-07,
"input_cost_per_token": 5e-06,
@ -35928,6 +35949,7 @@
},
"perplexity/anthropic/claude-opus-4-6": {
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"litellm_provider": "perplexity",
"mode": "responses",
"supports_web_search": true,
@ -39303,6 +39325,7 @@
},
"vercel_ai_gateway/anthropic/claude-opus-4.6": {
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 6.25e-06,
"cache_read_input_token_cost": 5e-07,
"input_cost_per_token": 5e-06,
@ -40562,6 +40585,7 @@
"deprecation_date": "2027-02-05",
"regional_endpoint_uplift_multiplier": 1.1,
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 6.25e-06,
"cache_creation_input_token_cost_above_1hr": 1e-05,
"cache_read_input_token_cost": 5e-07,
@ -40594,6 +40618,7 @@
"deprecation_date": "2027-02-05",
"regional_endpoint_uplift_multiplier": 1.1,
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 6.25e-06,
"cache_creation_input_token_cost_above_1hr": 1e-05,
"cache_read_input_token_cost": 5e-07,
@ -40959,6 +40984,7 @@
"vertex_ai/claude-sonnet-4-6": {
"regional_endpoint_uplift_multiplier": 1.1,
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 3.75e-06,
"cache_creation_input_token_cost_above_1hr": 6e-06,
"cache_read_input_token_cost": 3e-07,
@ -44042,10 +44068,10 @@
"comment": "5 credits per second @ $0.01 per credit = $0.05 per second"
}
},
"runwayml/gen4_aleph": {
"runwayml/gen4.5": {
"litellm_provider": "runwayml",
"mode": "video_generation",
"output_cost_per_video_per_second": 0.15,
"output_cost_per_second": 0.12,
"source": "https://docs.dev.runwayml.com/guides/pricing/",
"supported_modalities": [
"text",
@ -44055,13 +44081,136 @@
"video"
],
"metadata": {
"comment": "15 credits per second @ $0.01 per credit = $0.15 per second"
"comment": "12 credits per second @ $0.01 per credit = $0.12 per second"
}
},
"runwayml/gen3a_turbo": {
"runwayml/aleph2": {
"litellm_provider": "runwayml",
"mode": "video_generation",
"output_cost_per_video_per_second": 0.05,
"output_cost_per_second": 0.28,
"source": "https://docs.dev.runwayml.com/guides/pricing/",
"supported_modalities": [
"text",
"video"
],
"supported_output_modalities": [
"video"
],
"metadata": {
"comment": "28 credits per second @ $0.01 per credit = $0.28 per second; 56 credit minimum per task not modeled"
}
},
"runwayml/seedance2": {
"litellm_provider": "runwayml",
"mode": "video_generation",
"output_cost_per_second": 0.36,
"output_cost_per_second_1080p": 0.4,
"output_cost_per_second_4k": 1.5,
"source": "https://docs.dev.runwayml.com/guides/pricing/",
"supported_modalities": [
"text",
"image",
"video"
],
"supported_output_modalities": [
"video"
],
"metadata": {
"comment": "36 credits per second at 480p/720p, 40 at 1080p, 150 at 4K @ $0.01 per credit"
}
},
"runwayml/seedance2_fast": {
"litellm_provider": "runwayml",
"mode": "video_generation",
"output_cost_per_second": 0.29,
"source": "https://docs.dev.runwayml.com/guides/pricing/",
"supported_modalities": [
"text",
"image",
"video"
],
"supported_output_modalities": [
"video"
],
"metadata": {
"comment": "29 credits per second at 480p/720p @ $0.01 per credit = $0.29 per second"
}
},
"runwayml/seedance2_mini": {
"litellm_provider": "runwayml",
"mode": "video_generation",
"output_cost_per_second": 0.16,
"source": "https://docs.dev.runwayml.com/guides/pricing/",
"supported_modalities": [
"text",
"image",
"video"
],
"supported_output_modalities": [
"video"
],
"metadata": {
"comment": "16 credits per second @ $0.01 per credit = $0.16 per second; 64 credit minimum per task not modeled"
}
},
"runwayml/seedance2_5": {
"litellm_provider": "runwayml",
"mode": "video_generation",
"output_cost_per_second": 0.3,
"output_cost_per_second_480p": 0.2,
"output_cost_per_second_1080p": 0.68,
"source": "https://docs.dev.runwayml.com/guides/pricing/",
"supported_modalities": [
"text",
"image",
"video"
],
"supported_output_modalities": [
"video"
],
"metadata": {
"comment": "Output: 20/30/68 credits per second at 480p/720p/1080p @ $0.01 per credit; input video billed additionally at 10/15/34 credits per input second and the 80 credit minimum per task are not modeled"
}
},
"runwayml/hailuo3": {
"litellm_provider": "runwayml",
"mode": "video_generation",
"output_cost_per_second": 0.1,
"output_cost_per_second_1080p": 0.15,
"source": "https://docs.dev.runwayml.com/guides/pricing/",
"supported_modalities": [
"text",
"image",
"video"
],
"supported_output_modalities": [
"video"
],
"metadata": {
"comment": "10 credits per second at 768P, 15 at 2K (mapped to the 1080p tier) @ $0.01 per credit; 2 credits per reference image not modeled"
}
},
"runwayml/gemini_omni_flash": {
"litellm_provider": "runwayml",
"mode": "video_generation",
"output_cost_per_second": 0.1,
"source": "https://docs.dev.runwayml.com/guides/pricing/",
"supported_modalities": [
"text",
"image",
"video"
],
"supported_output_modalities": [
"video"
],
"metadata": {
"comment": "10 credits per second @ $0.01 per credit = $0.10 per second"
}
},
"runwayml/veo3.1": {
"litellm_provider": "runwayml",
"mode": "video_generation",
"output_cost_per_second": 0.4,
"source": "https://docs.dev.runwayml.com/guides/pricing/",
"supported_modalities": [
"text",
@ -44071,7 +44220,23 @@
"video"
],
"metadata": {
"comment": "5 credits per second @ $0.01 per credit = $0.05 per second"
"comment": "40 credits per second with audio, 20 without @ $0.01 per credit; priced at the with-audio rate"
}
},
"runwayml/veo3.1_fast": {
"litellm_provider": "runwayml",
"mode": "video_generation",
"output_cost_per_second": 0.15,
"source": "https://docs.dev.runwayml.com/guides/pricing/",
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"video"
],
"metadata": {
"comment": "15 credits per second with audio, 10 without @ $0.01 per credit; priced at the with-audio rate"
}
},
"runwayml/gen4_image": {
@ -48728,6 +48893,7 @@
"vertex_ai/claude-sonnet-4-6@default": {
"regional_endpoint_uplift_multiplier": 1.1,
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 3.75e-06,
"cache_creation_input_token_cost_above_1hr": 6e-06,
"cache_read_input_token_cost": 3e-07,
@ -49512,6 +49678,7 @@
},
"snowflake/claude-sonnet-4-6": {
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"max_tokens": 16384,
"max_input_tokens": 200000,
"max_output_tokens": 16384,
@ -50502,6 +50669,14 @@
"supports_adaptive_thinking": true
}
},
{
"name": "claude-legacy-thinking",
"pattern": "claude-[a-z]+-4[-._]6(?!\\d)",
"description": "Claude at version 4.6 exactly, in any id shape that contains claude-<family>-4-6 (dotted and underscored minors included, dated releases such as claude-sonnet-4-6-20260219 too). The 4.6 family is adaptive-thinking yet still accepts legacy thinking.type=enabled with budget_tokens, so the caller's hard budget cap is forwarded verbatim instead of being rewritten to an uncapped output_config.effort. The lookahead keeps two-digit minors such as 4-60 from matching. 4.7+ and 5+ majors reject the legacy shape and stay on the adaptive translation.",
"model_info": {
"supports_legacy_thinking": true
}
},
{
"name": "claude-always-on-thinking",
"pattern": "claude-(?:fable|mythos)-",

View file

@ -274,10 +274,8 @@ async def get_form_data(request: Request) -> dict[str, Any]:
Handles when OpenAI SDKs pass form keys as `timestamp_granularities[]="word"` instead of `timestamp_granularities=["word", "sentence"]`
"""
form: Final = await request.form()
form_data: Final = dict(form)
parsed_form_data: Final[dict[str, Any]] = {}
for key, value in form_data.items():
# OpenAI SDKs pass form keys as `timestamp_granularities[]="word"` instead of `timestamp_granularities=["word", "sentence"]`
for key, value in form.multi_items(): # not dict(form), which keeps only the last repeat
if key.endswith("[]"):
clean_key = key[:-2]
parsed_form_data.setdefault(clean_key, []).append(value)

View file

@ -433,6 +433,9 @@ def _update_litellm_params_for_health_check(model_info: dict, litellm_params: di
"""
Update the litellm params for health check.
- merges `model_info.health_check_params` into the probe request, so a deployment whose provider
requires a payload field litellm does not synthesize (e.g. `mediaSource` for Bedrock TwelveLabs
Pegasus) can supply it. The dedicated knobs below are applied afterwards and win on conflict.
- gets a short `messages` param for health check
- adds a bounded `max_tokens` when the deployment is a chat-style mode
(`chat`, `completion`, `responses`) or the operator explicitly opts in
@ -447,6 +450,16 @@ def _update_litellm_params_for_health_check(model_info: dict, litellm_params: di
model_info,
litellm_params, # any-ok: untyped router config dict
)
_health_check_params: Final = model_info.get("health_check_params", None)
if isinstance(_health_check_params, dict):
litellm_params.update(_health_check_params)
elif _health_check_params is not None:
logger.warning(
"health_check_params for model %s is a %s, expected a dict. Ignoring it.",
litellm_params.get("model"),
type(_health_check_params).__name__,
)
litellm_params["messages"] = _get_random_llm_message()
if _should_inject_health_check_max_tokens(
model_info,

View file

@ -1888,6 +1888,8 @@ async def test_model_connection(
# already resolved before reaching this endpoint; any remaining
# reference must have come from the request body.
_reject_os_environ_references(request_litellm_params)
if model_info:
_reject_os_environ_references(model_info)
model_name: Final = request_litellm_params.get("model")
# Look up model configuration from router if model name is provided
@ -1950,23 +1952,23 @@ async def test_model_connection(
**request_litellm_params,
}
## Auth check
auth_model_info: Final = loaded_model_info if loaded_model_info is not None else model_info
resolved_model_info: Final = loaded_model_info if loaded_model_info is not None else model_info
litellm_params = _update_litellm_params_for_health_check(
model_info=resolved_model_info or {},
litellm_params=litellm_params,
)
## Auth check, on the final probe params so health_check_params cannot retarget it afterwards
await ModelManagementAuthChecks.can_user_make_model_call(
model_params=Deployment(
model_name="test_model",
litellm_params=LiteLLM_Params(**litellm_params),
model_info=auth_model_info,
model_info=resolved_model_info,
),
user_api_key_dict=user_api_key_dict,
prisma_client=prisma_client,
premium_user=premium_user,
)
# Include health_check_params if provided
litellm_params = _update_litellm_params_for_health_check(
model_info={},
litellm_params=litellm_params,
)
mode = mode or litellm_params.pop("mode", None)
result: Final = await run_with_timeout(

View file

@ -2,7 +2,6 @@
from typing import Any, Final
import orjson
from fastapi import APIRouter, Depends, File, Form, Request, Response, UploadFile
from fastapi.responses import ORJSONResponse
@ -20,6 +19,7 @@ from litellm.proxy.video_endpoints.utils import (
encode_character_id_in_response,
extract_model_from_target_model_names,
get_custom_provider_from_data,
video_reference_to_id,
)
from litellm.types.videos.utils import (
decode_character_id_with_provider,
@ -451,9 +451,7 @@ async def video_remix(
version,
)
# Read request body
body: Final = await request.body()
data: Final = orjson.loads(body)
data: Final = await _read_request_body(request=request)
data["video_id"] = video_id
decoded: Final = decode_video_id_with_provider(video_id)
@ -760,15 +758,10 @@ async def video_edit(
version,
)
body: Final = await request.body()
data: Final = orjson.loads(body)
data: Final = await _read_request_body(request=request)
data["video_id"] = video_reference_to_id(data.pop("video", None))
# Extract video_id from nested video object
video_ref: Final = data.pop("video", {})
video_id: Final = video_ref.get("id", "") if isinstance(video_ref, dict) else ""
data["video_id"] = video_id
decoded: Final = decode_video_id_with_provider(video_id)
decoded: Final = decode_video_id_with_provider(data["video_id"])
provider_from_id: Final = decoded.get("custom_llm_provider")
model_id_from_decoded: Final = decoded.get("model_id")
@ -860,15 +853,10 @@ async def video_extension(
version,
)
body: Final = await request.body()
data: Final = orjson.loads(body)
data: Final = await _read_request_body(request=request)
data["video_id"] = video_reference_to_id(data.pop("video", None))
# Extract video_id from nested video object
video_ref: Final = data.pop("video", {})
video_id: Final = video_ref.get("id", "") if isinstance(video_ref, dict) else ""
data["video_id"] = video_id
decoded: Final = decode_video_id_with_provider(video_id)
decoded: Final = decode_video_id_with_provider(data["video_id"])
provider_from_id: Final = decoded.get("custom_llm_provider")
model_id_from_decoded: Final = decoded.get("model_id")

View file

@ -13,6 +13,18 @@ def extract_model_from_target_model_names(target_model_names: Any) -> str | None
return target_model_names[0] if target_model_names else None
def video_reference_to_id(video_ref: object) -> str:
if isinstance(video_ref, dict):
return video_ref.get("id", "")
if not isinstance(video_ref, str):
return ""
try:
parsed_ref: Final = orjson.loads(video_ref)
except orjson.JSONDecodeError:
return video_ref
return parsed_ref.get("id", "") if isinstance(parsed_ref, dict) else video_ref
def get_custom_provider_from_data(data: dict[str, Any]) -> str | None:
custom_llm_provider: Final = data.get("custom_llm_provider")
if custom_llm_provider:

View file

@ -10,7 +10,7 @@ from typing import Any, ClassVar, Final, Generic, Literal, TypeVar, get_type_hin
import httpx
from pydantic import BaseModel, ConfigDict, Field, field_validator, model_validator
from typing_extensions import Protocol, Required, TypedDict, runtime_checkable
from typing_extensions import Protocol, ReadOnly, Required, TypedDict, runtime_checkable
from litellm._uuid import uuid
@ -480,7 +480,9 @@ class LiteLLMParamsTypedDict(TypedDict, total=False):
output_cost_per_token: float | None
input_cost_per_second: float | None
output_cost_per_second: float | None
output_cost_per_second_480p: ReadOnly[float | None]
output_cost_per_second_1080p: float | None
output_cost_per_second_4k: ReadOnly[float | None]
num_retries: int | None
## MOCK RESPONSES ##
mock_response: str | ModelResponse | Exception | None

View file

@ -154,6 +154,7 @@ class ProviderSpecificModelInfo(TypedDict, total=False):
supports_web_search: bool | None
supports_reasoning: bool | None
supports_adaptive_thinking: bool | None
supports_legacy_thinking: ReadOnly[bool | None]
thinking_always_on: ReadOnly[bool | None]
supports_tool_search: bool | None
supports_mid_conversation_system: bool | None
@ -277,6 +278,8 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False):
output_cost_per_second_1080p: (
float | None
) # video_generation tier: key output_cost_per_second_<resolution> (e.g. 1080p, 720p)
output_cost_per_second_480p: ReadOnly[float | None]
output_cost_per_second_4k: ReadOnly[float | None]
ocr_cost_per_page: float | None # for OCR models
ocr_cost_per_credit: float | None # for OCR models priced by credit
annotation_cost_per_page: float | None # for OCR models
@ -3337,6 +3340,8 @@ class CustomPricingLiteLLMParams(MirroredPricingParams):
input_cost_per_second: float | None = None
output_cost_per_second: float | None = None
output_cost_per_second_1080p: float | None = None
output_cost_per_second_480p: float | None = None
output_cost_per_second_4k: float | None = None
input_cost_per_pixel: float | None = None
output_cost_per_pixel: float | None = None

View file

@ -5726,6 +5726,8 @@ def _get_model_info_helper(
),
output_cost_per_second=_model_info.get("output_cost_per_second", None),
output_cost_per_second_1080p=_model_info.get("output_cost_per_second_1080p", None),
output_cost_per_second_480p=_model_info.get("output_cost_per_second_480p", None),
output_cost_per_second_4k=_model_info.get("output_cost_per_second_4k", None),
output_cost_per_video_per_second=_model_info.get("output_cost_per_video_per_second", None),
output_cost_per_image=_model_info.get("output_cost_per_image", None),
output_cost_per_image_token=_model_info.get("output_cost_per_image_token", None),
@ -5753,6 +5755,7 @@ def _get_model_info_helper(
supports_url_context=_model_info.get("supports_url_context", None),
supports_reasoning=_model_info.get("supports_reasoning", None),
supports_adaptive_thinking=_model_info.get("supports_adaptive_thinking", None),
supports_legacy_thinking=_model_info.get("supports_legacy_thinking", None),
thinking_always_on=_model_info.get("thinking_always_on", None),
supports_tool_search=_model_info.get("supports_tool_search", None),
supports_mid_conversation_system=_model_info.get("supports_mid_conversation_system", None),
@ -6526,21 +6529,10 @@ def acreate(*args, **kwargs): ## Thin client to handle the acreate langchain ca
def prompt_token_calculator(model, messages):
# use tiktoken or anthropic's tokenizer depending on the model
text: Final = " ".join(message["content"] for message in messages)
num_tokens = 0
if "claude" in model:
try:
import anthropic
except Exception:
Exception("Anthropic import failed please run `pip install anthropic`")
from anthropic import AI_PROMPT, HUMAN_PROMPT, Anthropic
anthropic_obj: Final = Anthropic()
num_tokens = anthropic_obj.count_tokens(text)
else:
num_tokens = len(_get_default_encoding().encode(text))
return num_tokens
return token_counter(model=model, text=text)
return len(_get_default_encoding().encode(text))
def valid_model(model):

View file

@ -1019,6 +1019,7 @@
},
"anthropic.claude-opus-4-6-v1": {
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 6.25e-06,
"cache_creation_input_token_cost_above_1hr": 1e-05,
"cache_read_input_token_cost": 5e-07,
@ -1053,6 +1054,7 @@
},
"global.anthropic.claude-opus-4-6-v1": {
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 6.25e-06,
"cache_creation_input_token_cost_above_1hr": 1e-05,
"cache_read_input_token_cost": 5e-07,
@ -1087,6 +1089,7 @@
},
"us.anthropic.claude-opus-4-6-v1": {
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 6.875e-06,
"cache_creation_input_token_cost_above_1hr": 1.1e-05,
"cache_read_input_token_cost": 5.5e-07,
@ -1121,6 +1124,7 @@
},
"eu.anthropic.claude-opus-4-6-v1": {
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 6.875e-06,
"cache_creation_input_token_cost_above_1hr": 1.1e-05,
"cache_read_input_token_cost": 5.5e-07,
@ -1155,6 +1159,7 @@
},
"au.anthropic.claude-opus-4-6-v1": {
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 6.875e-06,
"cache_creation_input_token_cost_above_1hr": 1.1e-05,
"cache_read_input_token_cost": 5.5e-07,
@ -2233,6 +2238,7 @@
},
"anthropic.claude-sonnet-4-6": {
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 3.75e-06,
"cache_creation_input_token_cost_above_1hr": 6e-06,
"cache_read_input_token_cost": 3e-07,
@ -2266,6 +2272,7 @@
},
"global.anthropic.claude-sonnet-4-6": {
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 3.75e-06,
"cache_creation_input_token_cost_above_1hr": 6e-06,
"cache_read_input_token_cost": 3e-07,
@ -2299,6 +2306,7 @@
},
"us.anthropic.claude-sonnet-4-6": {
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 4.125e-06,
"cache_creation_input_token_cost_above_1hr": 6.6e-06,
"cache_read_input_token_cost": 3.3e-07,
@ -2332,6 +2340,7 @@
},
"eu.anthropic.claude-sonnet-4-6": {
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 4.125e-06,
"cache_creation_input_token_cost_above_1hr": 6.6e-06,
"cache_read_input_token_cost": 3.3e-07,
@ -2365,6 +2374,7 @@
},
"au.anthropic.claude-sonnet-4-6": {
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 4.125e-06,
"cache_creation_input_token_cost_above_1hr": 6.6e-06,
"cache_read_input_token_cost": 3.3e-07,
@ -2398,6 +2408,7 @@
},
"jp.anthropic.claude-sonnet-4-6": {
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 4.125e-06,
"cache_creation_input_token_cost_above_1hr": 6.6e-06,
"cache_read_input_token_cost": 3.3e-07,
@ -2950,6 +2961,7 @@
"azure_ai/claude-opus-4-6": {
"deprecation_date": "2027-02-02",
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"input_cost_per_token": 5e-06,
"output_cost_per_token": 2.5e-05,
"litellm_provider": "azure_ai",
@ -3181,6 +3193,7 @@
"azure_ai/claude-sonnet-4-6": {
"deprecation_date": "2027-02-10",
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 3.75e-06,
"cache_creation_input_token_cost_above_1hr": 6e-06,
"cache_read_input_token_cost": 3e-07,
@ -12489,6 +12502,7 @@
"search_context_size_medium": 0.01
},
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"supports_assistant_prefill": true,
"supports_computer_use": true,
"supports_function_calling": true,
@ -12698,6 +12712,7 @@
"search_context_size_medium": 0.01
},
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"supports_assistant_prefill": false,
"supports_computer_use": true,
"supports_function_calling": true,
@ -12735,6 +12750,7 @@
"search_context_size_medium": 0.01
},
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"supports_assistant_prefill": false,
"supports_computer_use": true,
"supports_function_calling": true,
@ -14724,6 +14740,7 @@
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_legacy_thinking": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_tool_choice": true
@ -14892,6 +14909,7 @@
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_legacy_thinking": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_tool_choice": true
@ -23427,6 +23445,7 @@
},
"github_copilot/claude-opus-4.6-fast": {
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"litellm_provider": "github_copilot",
"max_input_tokens": 128000,
"max_output_tokens": 16000,
@ -33810,6 +33829,7 @@
},
"openrouter/anthropic/claude-sonnet-4.6": {
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 3.75e-06,
"cache_creation_input_token_cost_above_200k_tokens": 7.5e-06,
"cache_read_input_token_cost": 3e-07,
@ -33854,6 +33874,7 @@
},
"openrouter/anthropic/claude-opus-4.6": {
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 6.25e-06,
"cache_read_input_token_cost": 5e-07,
"input_cost_per_token": 5e-06,
@ -35928,6 +35949,7 @@
},
"perplexity/anthropic/claude-opus-4-6": {
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"litellm_provider": "perplexity",
"mode": "responses",
"supports_web_search": true,
@ -39303,6 +39325,7 @@
},
"vercel_ai_gateway/anthropic/claude-opus-4.6": {
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 6.25e-06,
"cache_read_input_token_cost": 5e-07,
"input_cost_per_token": 5e-06,
@ -40562,6 +40585,7 @@
"deprecation_date": "2027-02-05",
"regional_endpoint_uplift_multiplier": 1.1,
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 6.25e-06,
"cache_creation_input_token_cost_above_1hr": 1e-05,
"cache_read_input_token_cost": 5e-07,
@ -40594,6 +40618,7 @@
"deprecation_date": "2027-02-05",
"regional_endpoint_uplift_multiplier": 1.1,
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 6.25e-06,
"cache_creation_input_token_cost_above_1hr": 1e-05,
"cache_read_input_token_cost": 5e-07,
@ -40959,6 +40984,7 @@
"vertex_ai/claude-sonnet-4-6": {
"regional_endpoint_uplift_multiplier": 1.1,
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 3.75e-06,
"cache_creation_input_token_cost_above_1hr": 6e-06,
"cache_read_input_token_cost": 3e-07,
@ -44042,10 +44068,10 @@
"comment": "5 credits per second @ $0.01 per credit = $0.05 per second"
}
},
"runwayml/gen4_aleph": {
"runwayml/gen4.5": {
"litellm_provider": "runwayml",
"mode": "video_generation",
"output_cost_per_video_per_second": 0.15,
"output_cost_per_second": 0.12,
"source": "https://docs.dev.runwayml.com/guides/pricing/",
"supported_modalities": [
"text",
@ -44055,13 +44081,136 @@
"video"
],
"metadata": {
"comment": "15 credits per second @ $0.01 per credit = $0.15 per second"
"comment": "12 credits per second @ $0.01 per credit = $0.12 per second"
}
},
"runwayml/gen3a_turbo": {
"runwayml/aleph2": {
"litellm_provider": "runwayml",
"mode": "video_generation",
"output_cost_per_video_per_second": 0.05,
"output_cost_per_second": 0.28,
"source": "https://docs.dev.runwayml.com/guides/pricing/",
"supported_modalities": [
"text",
"video"
],
"supported_output_modalities": [
"video"
],
"metadata": {
"comment": "28 credits per second @ $0.01 per credit = $0.28 per second; 56 credit minimum per task not modeled"
}
},
"runwayml/seedance2": {
"litellm_provider": "runwayml",
"mode": "video_generation",
"output_cost_per_second": 0.36,
"output_cost_per_second_1080p": 0.4,
"output_cost_per_second_4k": 1.5,
"source": "https://docs.dev.runwayml.com/guides/pricing/",
"supported_modalities": [
"text",
"image",
"video"
],
"supported_output_modalities": [
"video"
],
"metadata": {
"comment": "36 credits per second at 480p/720p, 40 at 1080p, 150 at 4K @ $0.01 per credit"
}
},
"runwayml/seedance2_fast": {
"litellm_provider": "runwayml",
"mode": "video_generation",
"output_cost_per_second": 0.29,
"source": "https://docs.dev.runwayml.com/guides/pricing/",
"supported_modalities": [
"text",
"image",
"video"
],
"supported_output_modalities": [
"video"
],
"metadata": {
"comment": "29 credits per second at 480p/720p @ $0.01 per credit = $0.29 per second"
}
},
"runwayml/seedance2_mini": {
"litellm_provider": "runwayml",
"mode": "video_generation",
"output_cost_per_second": 0.16,
"source": "https://docs.dev.runwayml.com/guides/pricing/",
"supported_modalities": [
"text",
"image",
"video"
],
"supported_output_modalities": [
"video"
],
"metadata": {
"comment": "16 credits per second @ $0.01 per credit = $0.16 per second; 64 credit minimum per task not modeled"
}
},
"runwayml/seedance2_5": {
"litellm_provider": "runwayml",
"mode": "video_generation",
"output_cost_per_second": 0.3,
"output_cost_per_second_480p": 0.2,
"output_cost_per_second_1080p": 0.68,
"source": "https://docs.dev.runwayml.com/guides/pricing/",
"supported_modalities": [
"text",
"image",
"video"
],
"supported_output_modalities": [
"video"
],
"metadata": {
"comment": "Output: 20/30/68 credits per second at 480p/720p/1080p @ $0.01 per credit; input video billed additionally at 10/15/34 credits per input second and the 80 credit minimum per task are not modeled"
}
},
"runwayml/hailuo3": {
"litellm_provider": "runwayml",
"mode": "video_generation",
"output_cost_per_second": 0.1,
"output_cost_per_second_1080p": 0.15,
"source": "https://docs.dev.runwayml.com/guides/pricing/",
"supported_modalities": [
"text",
"image",
"video"
],
"supported_output_modalities": [
"video"
],
"metadata": {
"comment": "10 credits per second at 768P, 15 at 2K (mapped to the 1080p tier) @ $0.01 per credit; 2 credits per reference image not modeled"
}
},
"runwayml/gemini_omni_flash": {
"litellm_provider": "runwayml",
"mode": "video_generation",
"output_cost_per_second": 0.1,
"source": "https://docs.dev.runwayml.com/guides/pricing/",
"supported_modalities": [
"text",
"image",
"video"
],
"supported_output_modalities": [
"video"
],
"metadata": {
"comment": "10 credits per second @ $0.01 per credit = $0.10 per second"
}
},
"runwayml/veo3.1": {
"litellm_provider": "runwayml",
"mode": "video_generation",
"output_cost_per_second": 0.4,
"source": "https://docs.dev.runwayml.com/guides/pricing/",
"supported_modalities": [
"text",
@ -44071,7 +44220,23 @@
"video"
],
"metadata": {
"comment": "5 credits per second @ $0.01 per credit = $0.05 per second"
"comment": "40 credits per second with audio, 20 without @ $0.01 per credit; priced at the with-audio rate"
}
},
"runwayml/veo3.1_fast": {
"litellm_provider": "runwayml",
"mode": "video_generation",
"output_cost_per_second": 0.15,
"source": "https://docs.dev.runwayml.com/guides/pricing/",
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"video"
],
"metadata": {
"comment": "15 credits per second with audio, 10 without @ $0.01 per credit; priced at the with-audio rate"
}
},
"runwayml/gen4_image": {
@ -48728,6 +48893,7 @@
"vertex_ai/claude-sonnet-4-6@default": {
"regional_endpoint_uplift_multiplier": 1.1,
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"cache_creation_input_token_cost": 3.75e-06,
"cache_creation_input_token_cost_above_1hr": 6e-06,
"cache_read_input_token_cost": 3e-07,
@ -49512,6 +49678,7 @@
},
"snowflake/claude-sonnet-4-6": {
"supports_adaptive_thinking": true,
"supports_legacy_thinking": true,
"max_tokens": 16384,
"max_input_tokens": 200000,
"max_output_tokens": 16384,
@ -50502,6 +50669,14 @@
"supports_adaptive_thinking": true
}
},
{
"name": "claude-legacy-thinking",
"pattern": "claude-[a-z]+-4[-._]6(?!\\d)",
"description": "Claude at version 4.6 exactly, in any id shape that contains claude-<family>-4-6 (dotted and underscored minors included, dated releases such as claude-sonnet-4-6-20260219 too). The 4.6 family is adaptive-thinking yet still accepts legacy thinking.type=enabled with budget_tokens, so the caller's hard budget cap is forwarded verbatim instead of being rewritten to an uncapped output_config.effort. The lookahead keeps two-digit minors such as 4-60 from matching. 4.7+ and 5+ majors reject the legacy shape and stay on the adaptive translation.",
"model_info": {
"supports_legacy_thinking": true
}
},
{
"name": "claude-always-on-thinking",
"pattern": "claude-(?:fable|mythos)-",

View file

@ -428,6 +428,14 @@
"type": "number",
"minimum": 0
},
"output_cost_per_second_480p": {
"type": "number",
"minimum": 0
},
"output_cost_per_second_4k": {
"type": "number",
"minimum": 0
},
"output_cost_per_token": {
"type": "number",
"minimum": 0,
@ -625,6 +633,9 @@
"supports_image_size": {
"type": "boolean"
},
"supports_legacy_thinking": {
"type": "boolean"
},
"supports_low_reasoning_effort": {
"type": "boolean"
},

View file

@ -57,7 +57,7 @@
"limit": 3
},
"BLE001": {
"limit": 2920
"limit": 2919
},
"C401": {
"limit": 8
@ -108,7 +108,7 @@
"limit": 3
},
"F401": {
"limit": 17
"limit": 14
},
"LOG015": {
"limit": 5
@ -152,9 +152,6 @@
"PLW0127": {
"limit": 57
},
"PLW0133": {
"limit": 1
},
"PLW0602": {
"limit": 215
},

View file

@ -40,6 +40,17 @@
# later binding makes the name local for the whole body, so the read raises
# UnboundLocalError, and in an autouse fixture that takes every test in the
# directory down with it
# F601 the same key literal twice in one dict. Python keeps the last value, so the
# first is dropped before the test ever runs, and a fixture that looks like it
# covers two cases covers one
# B023 a closure over a loop variable. Every closure sees the last iteration's value,
# so a per-case callback built in a loop checks the last case N times. Bind the
# value as a parameter instead
# B025 an `except` for a type an earlier `except` already catches. The second handler
# is unreachable, so the recovery or skip written there never happens
# F632 `is` against a literal. It compares identity, so it passes only where CPython
# happens to intern the value and stops meaning what it says the moment the
# value is built at runtime
#
# No target-version here on purpose: it resolves from requires-python (>=3.10), so
# 3.11-only builtins like BaseExceptionGroup are correctly flagged in a tree that
@ -63,4 +74,8 @@ lint.select = [
"PLW0127",
"RUF043",
"F823",
"F601",
"B023",
"B025",
"F632",
]

View file

@ -5,9 +5,9 @@ lint.ignore = ["F405", "E402", "F403"]
lint.extend-select = [
"T20", "PGH004", "RUF008", "RUF009", "RUF100",
"B033", "FURB136", "FURB168", "FURB188", "I001", "PERF402", "PIE790", "PIE800", "PLC0208",
"PLR0402", "PLR1711", "PLR1730", "PLR2044", "PYI030", "PYI041", "PYI064", "RET501", "RUF010",
"RUF022", "RUF023", "RUF051", "SIM114", "SIM118", "TC005", "UP006", "UP007", "UP008", "UP012",
"UP018", "UP024", "UP032", "UP034", "UP035", "UP037", "UP045",
"PLR0402", "PLR1711", "PLR1730", "PLR2044", "PLW0133", "PYI030", "PYI041", "PYI064", "RET501",
"RUF010", "RUF022", "RUF023", "RUF051", "SIM114", "SIM118", "TC005", "UP006", "UP007", "UP008",
"UP012", "UP018", "UP024", "UP032", "UP034", "UP035", "UP037", "UP045",
]
# RUF100 (unused-noqa) only knows the rules enabled in THIS config, so it would strip
# `# noqa` directives that protect rules enforced elsewhere. List those codes as external

View file

@ -93,7 +93,7 @@ def get_bedrock_pricing(url, providers):
else:
# General logic for other providers
section = soup.find(
"h2", text=lambda t: t and provider.lower() in t.lower()
"h2", text=lambda t, needle=provider.lower(): t and needle in t.lower()
)
if not section:
pricing_data[provider] = "Provider section not found"

View file

@ -64,11 +64,6 @@ def test_langsmith_logging_async():
except Exception as e:
pytest.fail(f"An exception occurred - {e}")
except litellm.Timeout as e:
pass
except Exception as e:
pytest.fail(f"An exception occurred - {e}")
async def make_async_calls(metadata=None, **completion_kwargs):
total_tasks = 300

View file

@ -4202,13 +4202,7 @@ def test_gemini_google_maps_tool_simple():
)
print(f"Response: {response.model_dump_json(indent=4)}")
assert response.choices[0].message.content is not None
except (litellm.RateLimitError, litellm.InternalServerError):
# Transient Vertex-side failures (rate limiting, 500 INTERNAL from the
# Google Maps grounding backend) are not LiteLLM bugs — don't fail CI.
pass
except litellm.InternalServerError:
pytest.skip(
"Google Maps Platform returned a transient 500 (upstream flake); skipping."
)
except (litellm.RateLimitError, litellm.InternalServerError) as e:
pytest.skip(f"Transient Vertex-side failure, not a LiteLLM bug: {e}")
except Exception as e:
pytest.fail(f"Error occurred: {e}")

View file

@ -143,7 +143,6 @@ def test_spend_logs_payload(model_id: Optional[str]):
"completion_start_time": datetime.datetime(2024, 6, 7, 12, 43, 30, 954146),
"max_tokens": 10,
"extra_body": {},
"custom_llm_provider": "azure",
"input": [
{"role": "system", "content": "you are a helpful assistant.\n"},
{"role": "user", "content": "bom dia"},

View file

@ -3323,16 +3323,17 @@ async def test_team_access_groups(prisma_client):
request._url = URL(url="/chat/completions")
def body_reader(requested_model: str):
async def return_body() -> bytes:
return f'{{"model": "{requested_model}"}}'.encode()
return return_body
for model in ["gpt-4o", "gemini-pro-vision"]:
# Expect these to pass
async def return_body():
return_string = f'{{"model": "{model}"}}'
# return string as bytes
return return_string.encode()
request = Request(scope={"type": "http"})
request._url = URL(url="/chat/completions")
request.body = return_body
request.body = body_reader(model)
# use generated key to auth in
print(
@ -3342,14 +3343,9 @@ async def test_team_access_groups(prisma_client):
for model in ["gpt-4", "gpt-4o-mini", "gemini-experimental"]:
# Expect these to fail
async def return_body_2():
return_string = f'{{"model": "{model}"}}'
# return string as bytes
return return_string.encode()
request = Request(scope={"type": "http"})
request._url = URL(url="/chat/completions")
request.body = return_body_2
request.body = body_reader(model)
# use generated key to auth in
print(

View file

@ -867,6 +867,7 @@ PROVIDERS_WITH_A_HANDLER = (
"openrouter",
"perplexity",
"replicate",
"runwayml",
"sagemaker",
"together_ai",
"vertex_ai",
@ -956,6 +957,7 @@ PROVIDERS_THAT_RECOGNISE_A_FULL_CONTEXT_WINDOW = (
"mistral",
"openai",
"perplexity",
"runwayml",
"together_ai",
"vertex_ai",
"xai",
@ -971,6 +973,7 @@ PROVIDERS_THAT_RECOGNISE_A_CONTENT_POLICY_BLOCK = (
"mistral",
"openai",
"perplexity",
"runwayml",
"together_ai",
"xai",
)

View file

@ -2,7 +2,6 @@
import pytest
import litellm
from litellm.constants import (
DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET,
DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET,
@ -17,7 +16,6 @@ from litellm.llms.bedrock.messages.invoke_transformations.anthropic_claude3_tran
)
@pytest.mark.parametrize(
"reasoning_effort,expected_effort",
[
@ -258,19 +256,22 @@ def test_reasoning_effort_in_supported_params():
"model",
[
"claude-sonnet-4-6",
"bedrock/invoke/us.anthropic.claude-sonnet-4-6",
"vertex_ai/claude-sonnet-4-6",
"claude-opus-4-6",
"claude-sonnet-4-6-20260219",
"bedrock/invoke/us.anthropic.claude-sonnet-4-6",
"bedrock/invoke/us.anthropic.claude-opus-4-6-v1:0",
"vertex_ai/claude-sonnet-4-6",
"vertex_ai/claude-opus-4-6",
"azure_ai/claude-sonnet-4-6",
],
)
def test_legacy_thinking_high_budget_clamps_to_high_when_xhigh_unsupported(
local_model_cost_map, model
):
"""Claude Code sends ``thinking.budget_tokens=31999``; Sonnet 4.6 and Opus 4.6
have no ``xhigh`` tier, so the translator must emit ``high`` rather than the
provider-invalid ``xhigh`` (regression for issue #29282)."""
def test_legacy_thinking_budget_preserved_verbatim_on_46(local_model_cost_map, model):
"""Regression for the passthrough silently dropping a caller's hard thinking
budget: the 4.6 family accepts ``thinking.type=enabled`` with ``budget_tokens``
natively, so rewriting it to ``thinking.type=adaptive`` + ``output_config.effort``
(which carries no ceiling) let reasoning run past the requested cap. The legacy
shape must be forwarded verbatim, in every 4.6 id shape including unmapped dated
releases resolved by the ``claude-legacy-thinking`` fallback rule."""
config = AnthropicMessagesConfig()
optional_params = {
"max_tokens": 1024,
@ -285,8 +286,8 @@ def test_legacy_thinking_high_budget_clamps_to_high_when_xhigh_unsupported(
headers={},
)
assert result.get("thinking") == {"type": "adaptive"}
assert result.get("output_config") == {"effort": "high"}
assert result.get("thinking") == {"type": "enabled", "budget_tokens": 31999}
assert "output_config" not in result
def test_legacy_thinking_high_budget_keeps_xhigh_when_supported():
@ -343,11 +344,44 @@ def test_legacy_thinking_translates_to_adaptive_for_opus_48(
assert result.get("output_config") == {"effort": "xhigh"}
@pytest.mark.parametrize(
"model,expected_effort",
[
("claude-sonnet-5", "xhigh"),
("claude-opus-5", "xhigh"),
("claude-newfamily-6", "high"),
],
)
def test_legacy_thinking_translates_to_adaptive_for_5_and_future_models(
local_model_cost_map, model, expected_effort
):
"""The 5 families reject ``thinking.type=enabled``, so the adaptive translation
stays the safe default for every adaptive model not flagged
``supports_legacy_thinking``, unmapped future ids included. An unmapped id
cannot prove ``xhigh`` support, so its high-budget bucket clamps to ``high``."""
config = AnthropicMessagesConfig()
optional_params = {
"max_tokens": 1024,
"thinking": {"type": "enabled", "budget_tokens": 31999},
}
result = config.transform_anthropic_messages_request(
model=model,
messages=[{"role": "user", "content": "Hello"}],
anthropic_messages_optional_request_params=optional_params,
litellm_params={},
headers={},
)
assert result.get("thinking") == {"type": "adaptive"}
assert result.get("output_config") == {"effort": expected_effort}
@pytest.mark.parametrize(
"budget_tokens,expected_effort",
[
(DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET * 2, "high"),
(DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET, "high"),
(DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET * 2, "xhigh"),
(DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET, "xhigh"),
(DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET, "high"),
(DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET - 1, "medium"),
(DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET, "medium"),
@ -355,7 +389,9 @@ def test_legacy_thinking_translates_to_adaptive_for_opus_48(
(1, "low"),
],
)
def test_legacy_thinking_budget_buckets_on_sonnet_46(budget_tokens, expected_effort):
def test_legacy_thinking_budget_buckets_on_opus_48(
local_model_cost_map, budget_tokens, expected_effort
):
config = AnthropicMessagesConfig()
optional_params = {
"max_tokens": 1024,
@ -363,7 +399,7 @@ def test_legacy_thinking_budget_buckets_on_sonnet_46(budget_tokens, expected_eff
}
result = config.transform_anthropic_messages_request(
model="claude-sonnet-4-6",
model="claude-opus-4-8",
messages=[{"role": "user", "content": "Hello"}],
anthropic_messages_optional_request_params=optional_params,
litellm_params={},
@ -373,7 +409,29 @@ def test_legacy_thinking_budget_buckets_on_sonnet_46(budget_tokens, expected_eff
assert result.get("output_config") == {"effort": expected_effort}
def test_legacy_thinking_does_not_override_explicit_output_config():
def test_legacy_thinking_does_not_override_explicit_output_config(local_model_cost_map):
config = AnthropicMessagesConfig()
optional_params = {
"max_tokens": 1024,
"thinking": {"type": "enabled", "budget_tokens": 31999},
"output_config": {"effort": "low"},
}
result = config.transform_anthropic_messages_request(
model="claude-opus-4-8",
messages=[{"role": "user", "content": "Hello"}],
anthropic_messages_optional_request_params=optional_params,
litellm_params={},
headers={},
)
assert result.get("thinking") == {"type": "adaptive"}
assert result.get("output_config") == {"effort": "low"}
def test_legacy_thinking_with_explicit_output_config_untouched_on_46(
local_model_cost_map,
):
config = AnthropicMessagesConfig()
optional_params = {
"max_tokens": 1024,
@ -389,6 +447,7 @@ def test_legacy_thinking_does_not_override_explicit_output_config():
headers={},
)
assert result.get("thinking") == {"type": "enabled", "budget_tokens": 31999}
assert result.get("output_config") == {"effort": "low"}

View file

@ -475,7 +475,7 @@ class TestOllamaTextCompletionResponseIterator:
assert isinstance(result, ModelResponseStream)
assert result.choices and result.choices[0].delta is not None
assert result.choices[0].delta.content == None
assert getattr(result.choices[0].delta, "reasoning_content", None) is ""
assert getattr(result.choices[0].delta, "reasoning_content", None) == ""
def test_chunk_parser_done_chunk(self):
"""Test that done chunks work correctly."""

View file

@ -7,7 +7,12 @@ from unittest.mock import Mock
import httpx
import pytest
from litellm.llms.runwayml.videos.transformation import RunwayMLVideoConfig
from litellm.llms.base_llm.chat.transformation import BaseLLMException
from litellm.llms.runwayml.videos.transformation import (
RunwayMLError,
RunwayMLVideoConfig,
_ratio_to_resolution,
)
from litellm.types.router import GenericLiteLLMParams
from litellm.types.videos.main import VideoObject
@ -49,6 +54,158 @@ class TestRunwayMLVideoTransformation:
# Validate URL has correct endpoint
assert url == "https://api.dev.runwayml.com/v1/image_to_video"
def test_transform_video_create_request_text_to_video(self):
"""A prompt-only request must hit /text_to_video, not /image_to_video."""
data, files, url = self.config.transform_video_create_request(
model="veo3.1",
prompt="A serene mountain lake at sunrise",
api_base="https://api.dev.runwayml.com/v1",
video_create_optional_request_params={"duration": 8, "ratio": "1280:720"},
litellm_params=GenericLiteLLMParams(),
headers={},
)
assert url == "https://api.dev.runwayml.com/v1/text_to_video"
assert "promptImage" not in data
assert data["promptText"] == "A serene mountain lake at sunrise"
def test_transform_video_create_request_video_to_video(self):
"""A promptVideo request must hit /video_to_video with promptImage stripped."""
data, files, url = self.config.transform_video_create_request(
model="aleph2",
prompt="Make it snow",
api_base="https://api.dev.runwayml.com/v1",
video_create_optional_request_params={
"promptVideo": "https://example.com/source.mp4",
"promptImage": "https://example.com/reference.png",
"ratio": "1280:720",
},
litellm_params=GenericLiteLLMParams(),
headers={},
)
assert url == "https://api.dev.runwayml.com/v1/video_to_video"
assert data["promptVideo"] == "https://example.com/source.mp4"
assert "promptImage" not in data
def test_transform_video_create_request_video_uri_routes_to_video_to_video(self):
_, _, url = self.config.transform_video_create_request(
model="aleph2",
prompt="Make it snow",
api_base="https://api.dev.runwayml.com/v1",
video_create_optional_request_params={"videoUri": "https://example.com/source.mp4"},
litellm_params=GenericLiteLLMParams(),
headers={},
)
assert url == "https://api.dev.runwayml.com/v1/video_to_video"
def test_status_progress_fraction_scales_to_percent(self):
"""Runway reports progress as a 0..1 float; VideoObject.progress is an int percent."""
mock_response = Mock(spec=httpx.Response)
mock_response.json.return_value = {
"id": "63fd0f13-f29d-4e58-99d3-1cb9efa14a5b",
"createdAt": "2025-11-11T21:48:50.448Z",
"status": "RUNNING",
"progress": 0.027,
}
result = self.config.transform_video_status_retrieve_response(
raw_response=mock_response,
logging_obj=self.mock_logging_obj,
custom_llm_provider="runwayml",
)
assert result.status == "in_progress"
assert result.progress == 3
def test_status_progress_null_leaves_progress_unset(self):
"""Runway sends an explicit null progress for pending polls; scaling it must not crash."""
mock_response = Mock(spec=httpx.Response)
mock_response.json.return_value = {
"id": "63fd0f13-f29d-4e58-99d3-1cb9efa14a5b",
"createdAt": "2025-11-11T21:48:50.448Z",
"status": "PENDING",
"progress": None,
}
result = self.config.transform_video_status_retrieve_response(
raw_response=mock_response,
logging_obj=self.mock_logging_obj,
custom_llm_provider="runwayml",
)
assert result.status == "queued"
assert result.progress is None
def test_get_error_class_returns_exception_instead_of_raising(self):
error = self.config.get_error_class(
error_message="Invalid API key",
status_code=401,
headers={},
)
assert isinstance(error, RunwayMLError)
assert isinstance(error, BaseLLMException)
assert error.status_code == 401
assert error.message == "Invalid API key"
def test_create_response_usage_includes_resolution_and_provider_cost(self):
mock_response = Mock(spec=httpx.Response)
mock_response.json.return_value = {
"id": "test-video-id-123",
"createdAt": "2025-11-11T21:48:50.448Z",
"status": "PENDING",
"estimatedCost": {"credits": 25.0},
}
video_obj = self.config.transform_video_create_response(
model="gen4_turbo",
raw_response=mock_response,
logging_obj=self.mock_logging_obj,
custom_llm_provider="runwayml",
request_data={"model": "gen4_turbo", "ratio": "1280:720", "duration": 5},
)
assert video_obj.usage == {
"duration_seconds": 5.0,
"video_resolution": "720p",
"provider_reported_cost_usd": 0.25,
}
def test_create_response_usage_omits_unknown_fields(self):
mock_response = Mock(spec=httpx.Response)
mock_response.json.return_value = {
"id": "test-video-id-123",
"createdAt": "2025-11-11T21:48:50.448Z",
"status": "PENDING",
}
video_obj = self.config.transform_video_create_response(
model="gen4_turbo",
raw_response=mock_response,
logging_obj=self.mock_logging_obj,
custom_llm_provider="runwayml",
request_data={"model": "gen4_turbo"},
)
assert video_obj.usage == {}
@pytest.mark.parametrize(
"ratio,expected",
[
("848:480", "480p"),
("1280:720", "720p"),
("1920:1080", "1080p"),
("2560:1440", "1080p"),
("3840:2160", "4k"),
(None, None),
("banana", None),
],
)
def test_ratio_to_resolution_tiers(self, ratio, expected):
assert _ratio_to_resolution(ratio) == expected
def test_transform_video_status_with_timestamp_handling(self):
"""Test status retrieval handles RunwayML's ISO 8601 timestamps correctly."""
from litellm.types.videos.utils import encode_video_id_with_provider

View file

@ -124,11 +124,7 @@ class TestGenerateIAMToken:
mock_client.reset_mock()
mock_cache.reset_mock()
# Configure mock to return values based on env_keys
def get_secret_side_effect(key):
return env_keys.get(key)
mock_get_secret_str.side_effect = get_secret_side_effect
mock_get_secret_str.side_effect = env_keys.get
mock_response = MagicMock()
mock_response.json.return_value = {

View file

@ -18,6 +18,7 @@ from unittest.mock import AsyncMock, MagicMock
import orjson
import pytest
from starlette.datastructures import FormData
from litellm.ocr.main import convert_file_document_to_url_document, get_mime_type
@ -470,7 +471,7 @@ class TestProxySecurityGuard:
mock_request = MagicMock()
mock_request.headers = {"content-type": "multipart/form-data; boundary=---"}
mock_request.form = AsyncMock(return_value=mock_form)
mock_request.form = AsyncMock(return_value=FormData(mock_form))
result = await self._parse_multipart(mock_request)

View file

@ -5,6 +5,7 @@ import orjson
import pytest
from fastapi import Request
from fastapi.testclient import TestClient
from starlette.datastructures import FormData
@ -68,7 +69,7 @@ async def test_form_data_parsing():
test_data = {"name": "test_user", "message": "hello world"}
# Mock the form method to return the test data as an awaitable
mock_request.form = AsyncMock(return_value=test_data)
mock_request.form = AsyncMock(return_value=FormData(test_data))
mock_request.headers = {"content-type": "application/x-www-form-urlencoded"}
mock_request.scope = {}
mock_request.state._cached_headers = None
@ -119,7 +120,7 @@ async def test_form_data_with_json_metadata():
}
# Mock the form method to return the test data as an awaitable
mock_request.form = AsyncMock(return_value=test_data)
mock_request.form = AsyncMock(return_value=FormData(test_data))
mock_request.headers = {"content-type": "multipart/form-data"}
mock_request.scope = {}
mock_request.state._cached_headers = None
@ -160,7 +161,7 @@ async def test_form_data_with_invalid_json_metadata():
}
# Mock the form method to return the test data
mock_request.form = AsyncMock(return_value=test_data)
mock_request.form = AsyncMock(return_value=FormData(test_data))
mock_request.headers = {"content-type": "multipart/form-data"}
mock_request.scope = {}
mock_request.state._cached_headers = None
@ -183,7 +184,7 @@ async def test_form_data_without_metadata():
test_data = {"model": "whisper-1", "file": "audio.mp3", "language": "en"}
# Mock the form method to return the test data
mock_request.form = AsyncMock(return_value=test_data)
mock_request.form = AsyncMock(return_value=FormData(test_data))
mock_request.headers = {"content-type": "application/x-www-form-urlencoded"}
mock_request.scope = {}
mock_request.state._cached_headers = None
@ -214,7 +215,7 @@ async def test_form_data_with_empty_metadata():
}
# Mock the form method to return the test data
mock_request.form = AsyncMock(return_value=test_data)
mock_request.form = AsyncMock(return_value=FormData(test_data))
mock_request.headers = {"content-type": "multipart/form-data"}
mock_request.scope = {}
mock_request.state._cached_headers = None
@ -249,7 +250,7 @@ async def test_form_data_with_dict_metadata():
}
# Mock the form method to return the test data
mock_request.form = AsyncMock(return_value=test_data)
mock_request.form = AsyncMock(return_value=FormData(test_data))
mock_request.headers = {"content-type": "multipart/form-data"}
mock_request.scope = {}
mock_request.state._cached_headers = None
@ -280,7 +281,7 @@ async def test_form_data_with_none_metadata():
}
# Mock the form method to return the test data
mock_request.form = AsyncMock(return_value=test_data)
mock_request.form = AsyncMock(return_value=FormData(test_data))
mock_request.headers = {"content-type": "multipart/form-data"}
mock_request.scope = {}
mock_request.state._cached_headers = None
@ -495,33 +496,29 @@ async def test_surrogate_repair_skipped_above_size_limit(monkeypatch):
@pytest.mark.asyncio
async def test_get_form_data():
"""
Test that get_form_data correctly handles form data with array notation.
Tests audio transcription parameters as a specific example.
A repeated `foo[]` key is how the OpenAI SDKs send a list, so every value has to
survive. `FormData`, not a dict: a dict cannot even hold the duplicate key.
"""
# Create a mock request with transcription form data
mock_request = MagicMock()
mock_request.form = AsyncMock(
return_value=FormData(
[
("file", "file_object"),
("model", "gpt-4o-transcribe"),
("include[]", "logprobs"),
("language", "en"),
("prompt", "Transcribe this audio file"),
("response_format", "json"),
("stream", "false"),
("temperature", "0.2"),
("timestamp_granularities[]", "word"),
("timestamp_granularities[]", "segment"),
]
)
)
# Create mock form data with array notation for timestamp_granularities
mock_form_data = {
"file": "file_object", # In a real request this would be an UploadFile
"model": "gpt-4o-transcribe",
"include[]": "logprobs", # Array notation
"language": "en",
"prompt": "Transcribe this audio file",
"response_format": "json",
"stream": "false",
"temperature": "0.2",
"timestamp_granularities[]": "word", # First array item
"timestamp_granularities[]": "segment", # Second array item (would overwrite in dict, but handled by the function)
}
# Mock the form method to return the test data
mock_request.form = AsyncMock(return_value=mock_form_data)
# Call the function being tested
result = await get_form_data(mock_request)
# Verify regular form fields are preserved
assert result["file"] == "file_object"
assert result["model"] == "gpt-4o-transcribe"
assert result["language"] == "en"
@ -529,17 +526,8 @@ async def test_get_form_data():
assert result["response_format"] == "json"
assert result["stream"] == "false"
assert result["temperature"] == "0.2"
# Verify array fields are correctly parsed
assert "include" in result
assert isinstance(result["include"], list)
assert "logprobs" in result["include"]
assert "timestamp_granularities" in result
assert isinstance(result["timestamp_granularities"], list)
# Note: In a real MultiDict, both values would be present
# But in our mock dictionary the second value overwrites the first
assert "segment" in result["timestamp_granularities"]
assert result["include"] == ["logprobs"]
assert result["timestamp_granularities"] == ["word", "segment"]
def test_get_tags_from_request_body_with_metadata_tags():
@ -953,7 +941,7 @@ class TestReadRequestBodyNonCanonicalContentType:
mock_request = MagicMock()
mock_request.body = AsyncMock(return_value=orjson.dumps(payload))
mock_request.form = AsyncMock(return_value={})
mock_request.form = AsyncMock(return_value=FormData({}))
mock_request.headers = {"content-type": content_type}
mock_request.scope = {}
@ -964,7 +952,7 @@ class TestReadRequestBodyNonCanonicalContentType:
@pytest.mark.asyncio
async def test_real_form_post_still_parsed_as_form(self):
mock_request = MagicMock()
mock_request.form = AsyncMock(return_value={"k": "v"})
mock_request.form = AsyncMock(return_value=FormData({"k": "v"}))
mock_request.body = AsyncMock(return_value=b"")
mock_request.headers = {"content-type": "application/x-www-form-urlencoded"}
mock_request.scope = {}
@ -1020,7 +1008,7 @@ class TestGetRequestBody:
mock_request = MagicMock()
mock_request.method = "POST"
mock_request.headers = {"content-type": "multipart/form-data; boundary=x"}
mock_request.form = AsyncMock(return_value={"k": "v"})
mock_request.form = AsyncMock(return_value=FormData({"k": "v"}))
mock_request.scope = {}
result = await get_request_body(mock_request)

View file

@ -898,6 +898,62 @@ async def test_test_model_connection_uses_loaded_deployment_team_id_via_model_na
assert passed_model_params.model_info.team_id == deployment_owner_team_id
@pytest.mark.asyncio
async def test_test_model_connection_authorizes_on_params_after_health_check_params_merge():
"""
Regression guard for the ordering fix: health_check_params from the request
body are merged into the probe params BEFORE the authorization check, so a
caller cannot smuggle a field past auth via health_check_params. Auth is
stubbed to reject, which halts the endpoint right after it records the
params it was handed, so the outbound probe is never reached. If the merge
is moved back to after can_user_make_model_call, the marker is absent from
those params and this test fails.
"""
from fastapi import HTTPException
from litellm.proxy.management_endpoints.model_management_endpoints import (
ModelManagementAuthChecks,
)
from litellm.types.router import Deployment
marker = "sentinel-from-health-check-params"
mock_can_user_make_model_call = AsyncMock(
side_effect=HTTPException(status_code=403, detail="denied")
)
with (
patch( # test-quality-ok: proxy module global, no injection seam
"litellm.proxy.proxy_server.prisma_client", MagicMock()
),
patch( # test-quality-ok: proxy module global, no injection seam
"litellm.proxy.proxy_server.llm_router", None
),
patch.object( # test-quality-ok: capturing the params handed to auth is the assertion
ModelManagementAuthChecks,
"can_user_make_model_call",
mock_can_user_make_model_call,
),
pytest.raises(HTTPException),
):
await health_test_model_connection(
request=MagicMock(),
mode="chat",
litellm_params={"model": "openai/gpt-4o"},
model_info={"health_check_params": {"probe_marker": marker}},
user_api_key_dict=UserAPIKeyAuth(
token="requester-token",
user_id="admin-user",
user_role=LitellmUserRoles.PROXY_ADMIN,
),
)
assert mock_can_user_make_model_call.called
passed_model_params = mock_can_user_make_model_call.call_args.kwargs["model_params"]
assert isinstance(passed_model_params, Deployment)
authorized_params = passed_model_params.litellm_params.model_dump()
assert authorized_params.get("probe_marker") == marker
@pytest.mark.asyncio
async def test_test_model_connection_authorized_team_admin_passes_real_auth():
"""

View file

@ -12,6 +12,7 @@ import httpx
import pytest
from fastapi import HTTPException, Request, Response
from fastapi.testclient import TestClient
from starlette.datastructures import FormData
import litellm
@ -1380,7 +1381,7 @@ async def test_is_streaming_request_fn():
mock_request = Mock()
mock_request.method = "POST"
mock_request.headers = {"content-type": "multipart/form-data"}
mock_request.form = AsyncMock(return_value={"stream": "true"})
mock_request.form = AsyncMock(return_value=FormData({"stream": "true"}))
assert await is_streaming_request_fn(mock_request) is True

View file

@ -1,7 +1,11 @@
import json
import logging
from unittest.mock import AsyncMock, MagicMock, patch
import pytest
import respx
import litellm
from litellm.litellm_core_utils.health_check_helpers import HealthCheckHelpers
from litellm.proxy import health_check as hc_module
from litellm.proxy.health_check import (
@ -534,6 +538,135 @@ async def test_run_model_health_check_skips_auto_router_deployment():
assert result == {}
def test_health_check_params_merge_into_probe_params():
"""health_check_params reach the probe request for the deployment that declares them."""
media_source = {"s3Location": {"uri": "s3://my-bucket/clip.mp4"}}
updated = _update_litellm_params_for_health_check(
{"mode": "chat", "health_check_params": {"mediaSource": media_source}},
{"model": "bedrock/us.twelvelabs.pegasus-1-2-v1:0"},
)
assert updated["mediaSource"] == media_source
assert updated["model"] == "us.twelvelabs.pegasus-1-2-v1:0"
assert updated["custom_llm_provider"] == "bedrock"
def test_health_check_params_lose_to_dedicated_health_check_knobs():
"""The dedicated knobs are applied after the merge, so they win on conflict."""
model_info = {
"mode": "chat",
"health_check_params": {
"max_tokens": 4096,
"model": "openai/expensive-model",
"messages": [{"role": "user", "content": "from health_check_params"}],
"reasoning_effort": "high",
},
"health_check_max_tokens": 5,
"health_check_model": "openai/cheap-model",
"health_check_reasoning_effort": "none",
}
updated = _update_litellm_params_for_health_check(model_info, {"model": "openai/dummy"})
assert updated["max_tokens"] == 5
assert updated["model"] == "openai/cheap-model"
assert updated["reasoning_effort"] == "none"
assert updated["messages"] != model_info["health_check_params"]["messages"]
def test_health_check_params_lose_to_the_audio_speech_voice_knob():
"""health_check_voice still wins for audio_speech deployments."""
updated = _update_litellm_params_for_health_check(
{
"mode": "audio_speech",
"health_check_params": {"voice": "sage", "response_format": "wav"},
"health_check_voice": "shimmer",
},
{"model": "openai/tts-1"},
)
assert updated["voice"] == "shimmer"
assert updated["response_format"] == "wav"
@pytest.mark.parametrize(
"bad_value",
["mediaSource", ["mediaSource"], 5, True],
)
def test_health_check_params_ignored_when_not_a_dict(bad_value, caplog):
"""A misconfigured health_check_params is skipped with a warning instead of breaking the probe."""
with caplog.at_level(logging.WARNING, logger="litellm.proxy.health_check"):
updated = _update_litellm_params_for_health_check(
{"mode": "chat", "health_check_params": bad_value},
{"model": "openai/dummy"},
)
assert updated["model"] == "openai/dummy"
assert updated["max_tokens"] == 16
assert "health_check_params" in caplog.text
def test_health_check_params_apply_to_non_chat_modes():
"""Non-chat probes get health_check_params too, and still no max_tokens."""
updated = _update_litellm_params_for_health_check(
{"mode": "embedding", "health_check_params": {"dimensions": 8}},
{"model": "bedrock/amazon.titan-embed-text-v2:0"},
)
assert updated["dimensions"] == 8
assert "max_tokens" not in updated
async def _pegasus_health_check_request_body(
model_info: dict[str, object], monkeypatch: pytest.MonkeyPatch
) -> dict[str, object]:
monkeypatch.setattr(litellm, "disable_aiohttp_transport", True)
litellm.in_memory_llm_clients_cache.flush_cache()
litellm_params = _update_litellm_params_for_health_check(
model_info,
{
"model": "bedrock/us.twelvelabs.pegasus-1-2-v1:0",
"aws_access_key_id": "fake-access-key",
"aws_secret_access_key": "fake-secret-key",
"aws_region_name": "us-east-1",
},
)
with respx.mock(assert_all_called=True) as respx_mock:
invoke_route = respx_mock.post(
host="bedrock-runtime.us-east-1.amazonaws.com",
path__regex=r"/model/.+/invoke",
).respond(json={"message": "a person walks a dog", "finishReason": "stop"})
result = await litellm.ahealth_check(litellm_params, mode="chat")
assert "error" not in result, result
return json.loads(invoke_route.calls.last.request.content)
@pytest.mark.asyncio
async def test_health_check_params_reach_the_bedrock_invoke_body(monkeypatch):
"""The probe Bedrock actually receives carries mediaSource, which is what unblocks Pegasus."""
media_source = {"s3Location": {"uri": "s3://my-bucket/clip.mp4"}}
body = await _pegasus_health_check_request_body(
{"mode": "chat", "health_check_params": {"mediaSource": media_source}}, monkeypatch
)
assert body["mediaSource"] == media_source
assert body["maxOutputTokens"] == 16
assert body["inputPrompt"]
@pytest.mark.asyncio
async def test_bedrock_invoke_body_has_no_media_source_without_health_check_params(monkeypatch):
"""Negative control: the field only appears because the deployment asked for it."""
body = await _pegasus_health_check_request_body({"mode": "chat"}, monkeypatch)
assert "mediaSource" not in body
@pytest.mark.asyncio
async def test_run_model_health_check_skips_complexity_router_deployment():
fake_ahealth_check = AsyncMock(return_value={})

View file

@ -372,6 +372,7 @@ async def test_content__model_encoded_id(harness):
async def call_edit(
harness: Harness, *, body: Dict[str, Any], headers=None, query=None
):
harness.read_body.return_value = dict(body)
return await endpoints.video_edit(
request=FakeRequest(headers=headers, query=query, raw_body=orjson.dumps(body)),
fastapi_response=Response(),
@ -428,6 +429,27 @@ async def test_edit__missing_video_object_defaults_to_openai(harness):
assert "video" not in data
@pytest.mark.asyncio
async def test_edit__bare_string_video_id_from_form_field(harness):
await call_edit(harness, body={"prompt": "brighter", "video": "video_plain"})
assert harness.processor_data() == {
"prompt": "brighter",
"video_id": "video_plain",
"custom_llm_provider": "openai",
}
@pytest.mark.asyncio
async def test_edit__json_string_video_reference_from_form_field(harness):
await call_edit(
harness,
body={"prompt": "brighter", "video": orjson.dumps({"id": "video_plain"}).decode()},
)
assert harness.processor_data()["video_id"] == "video_plain"
# =========================================================================== #
# GET /v1/videos - video_list #
# =========================================================================== #
@ -471,6 +493,7 @@ async def test_list__provider_from_header(harness):
async def call_remix(
harness: Harness, video_id: str, *, body, headers=None, query=None
):
harness.read_body.return_value = dict(body)
return await endpoints.video_remix(
video_id=video_id,
request=FakeRequest(headers=headers, query=query, raw_body=orjson.dumps(body)),
@ -629,6 +652,7 @@ async def test_get_character__plain_id_defaults_openai_no_encode(harness):
async def call_extension(harness: Harness, *, body, headers=None, query=None):
harness.read_body.return_value = dict(body)
return await endpoints.video_extension(
request=FakeRequest(headers=headers, query=query, raw_body=orjson.dumps(body)),
fastapi_response=Response(),

View file

@ -1,8 +1,9 @@
"""
Pure-logic contract tests for litellm/proxy/video_endpoints/utils.py
Three helpers the video proxy endpoints lean on:
Four helpers the video proxy endpoints lean on:
- extract_model_from_target_model_names: first model from a comma string / list
- video_reference_to_id: normalize a video reference (dict / bare id / JSON string) to an id
- get_custom_provider_from_data: provider precedence (top-level > extra_body)
- encode_character_id_in_response: re-encode a response id in place
@ -20,6 +21,7 @@ from litellm.proxy.video_endpoints.utils import (
encode_character_id_in_response,
extract_model_from_target_model_names,
get_custom_provider_from_data,
video_reference_to_id,
)
from litellm.types.videos.utils import (
decode_character_id_with_provider,
@ -53,6 +55,31 @@ def test_extract_model__non_str_non_list_is_none(value):
assert extract_model_from_target_model_names(value) is None
# =========================================================================== #
# video_reference_to_id
# =========================================================================== #
@pytest.mark.parametrize(
"video_ref,expected",
[
({"id": "video_123"}, "video_123"), # dict reference -> its id
({"id": ""}, ""), # dict with empty id
({}, ""), # dict missing id -> default empty
({"other": "x"}, ""), # dict without id key
("video_123", "video_123"), # bare id string (not valid JSON) -> itself
('{"id": "video_9"}', "video_9"), # JSON-encoded dict -> its id
('{"other": 1}', ""), # JSON-encoded dict without id -> empty
("[1, 2]", "[1, 2]"), # JSON parses to non-dict -> original string
(None, ""), # non-str, non-dict
(123, ""), # non-str, non-dict
(["video_123"], ""), # list is neither dict nor str
],
)
def test_video_reference_to_id(video_ref, expected):
assert video_reference_to_id(video_ref) == expected
# =========================================================================== #
# get_custom_provider_from_data
# =========================================================================== #

View file

@ -1,6 +1,7 @@
import json
import logging
import os
import sys
from typing import Final
from unittest.mock import AsyncMock, MagicMock, patch
@ -41,6 +42,7 @@ from litellm.utils import (
get_prompt_cache_min_tokens,
is_cached_message,
is_prompt_caching_valid_prompt,
prompt_token_calculator,
)
# Adds the parent directory to the system path
@ -730,7 +732,9 @@ def validate_model_cost_values(model_data, exceptions=None):
"output_cost_per_pixel",
"input_cost_per_second",
"output_cost_per_second",
"output_cost_per_second_480p",
"output_cost_per_second_1080p",
"output_cost_per_second_4k",
"input_cost_per_query",
"input_cost_per_request",
"input_cost_per_audio_token",
@ -860,7 +864,6 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
"input_cost_per_character_above_128k_tokens": {"type": "number"},
"input_cost_per_image": {"type": "number"},
"input_cost_per_image_above_128k_tokens": {"type": "number"},
"input_cost_per_image_token": {"type": "number"},
"input_cost_per_video_token": {"type": "number"},
"input_cost_per_token_above_200k_tokens": {"type": "number"},
"input_cost_per_token_above_256k_tokens": {"type": "number"},
@ -944,7 +947,9 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
"output_cost_per_video_token": {"type": "number"},
"output_cost_per_pixel": {"type": "number"},
"output_cost_per_second": {"type": "number"},
"output_cost_per_second_480p": {"type": "number"},
"output_cost_per_second_1080p": {"type": "number"},
"output_cost_per_second_4k": {"type": "number"},
"output_cost_per_token": {"type": "number"},
"output_cost_per_token_above_128k_tokens": {"type": "number"},
"output_cost_per_token_above_200k_tokens": {"type": "number"},
@ -993,6 +998,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
"supports_xhigh_reasoning_effort": {"type": "boolean"},
"supports_max_reasoning_effort": {"type": "boolean"},
"supports_adaptive_thinking": {"type": "boolean"},
"supports_legacy_thinking": {"type": "boolean"},
"thinking_always_on": {"type": "boolean"},
"supports_mid_conversation_system": {"type": "boolean"},
"supports_sampling_params": {"type": "boolean"},
@ -1004,7 +1010,6 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
},
"bedrock_converse_supports_strict_tools": {"type": "boolean"},
"tpm": {"type": "number"},
"provider_specific_entry": {"type": "object"},
"supported_endpoints": {
"type": "array",
"items": {
@ -1124,6 +1129,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
exceptions = [
# Add any model IDs that should be exempt from the cost validation
# Example: "expensive-model-id",
"runwayml/seedance2", # 4K output is 150 credits/second = $1.50/second
]
is_valid, violations = validate_model_cost_values(actual_json, exceptions)
@ -4973,3 +4979,19 @@ def test_completion_does_not_leak_rust_flag_into_provider_request_body():
create_kwargs = mock_client.chat.completions.with_raw_response.create.call_args.kwargs
assert "rust" not in create_kwargs
assert "rust" not in (create_kwargs.get("extra_body") or {})
def test_prompt_token_calculator_counts_claude_without_the_anthropic_sdk():
"""
The claude branch used to call the anthropic SDK's `count_tokens`, which the SDK
removed, so every claude call raised AttributeError. Counting must work with
`anthropic` unimportable.
"""
messages: Final = [{"role": "user", "content": "the quick brown fox jumps over the lazy dog"}]
with patch.dict(sys.modules, {"anthropic": None}):
claude_tokens = prompt_token_calculator("claude-sonnet-4-5", messages)
gpt_tokens = prompt_token_calculator("gpt-4o", messages)
assert claude_tokens == 9
assert gpt_tokens == 9

View file

@ -421,6 +421,117 @@ class TestVideoGeneration:
)
assert cost == 0.5
def test_completion_cost_video_custom_pricing_under_litellm_metadata(self):
"""Video routes store deployment model_info under litellm_metadata, not metadata.
Regression for https://github.com/BerriAI/litellm/issues/36483: custom video
pricing was silently ignored because completion_cost only read metadata.
"""
from litellm.cost_calculator import completion_cost
mock_response = MagicMock()
mock_response.usage = {"duration_seconds": 10.0}
type(mock_response)._hidden_params = {}
mock_logging_obj = MagicMock()
mock_logging_obj.litellm_params = {
"litellm_metadata": {
"model_info": {
"output_cost_per_video_per_second": 0.18,
}
}
}
cost = completion_cost(
completion_response=mock_response,
model="runwayml/seedance2",
call_type="create_video",
custom_llm_provider="runwayml",
custom_pricing=True,
litellm_logging_obj=mock_logging_obj,
)
assert abs(cost - 1.8) < 0.001
def test_completion_cost_video_uses_provider_reported_cost_without_custom_pricing(self):
"""With no custom pricing, the provider's own reported cost wins over a duration estimate."""
from litellm.cost_calculator import completion_cost
mock_response = MagicMock()
mock_response.usage = {
"duration_seconds": 5.0,
"video_resolution": "720p",
"provider_reported_cost_usd": 0.31,
}
type(mock_response)._hidden_params = {}
cost = completion_cost(
completion_response=mock_response,
model="runwayml/gen4_turbo",
call_type="create_video",
custom_llm_provider="runwayml",
)
assert cost == 0.31
def test_completion_cost_video_custom_pricing_beats_provider_reported_cost(self):
"""Deployment-level custom pricing overrides the provider's reported cost."""
from litellm.cost_calculator import completion_cost
mock_response = MagicMock()
mock_response.usage = {
"duration_seconds": 10.0,
"provider_reported_cost_usd": 0.31,
}
type(mock_response)._hidden_params = {}
mock_logging_obj = MagicMock()
mock_logging_obj.litellm_params = {
"metadata": {
"model_info": {
"output_cost_per_video_per_second": 0.18,
}
}
}
cost = completion_cost(
completion_response=mock_response,
model="runwayml/seedance2",
call_type="create_video",
custom_llm_provider="runwayml",
custom_pricing=True,
litellm_logging_obj=mock_logging_obj,
)
assert abs(cost - 1.8) < 0.001
def test_completion_cost_video_resolution_tiers_from_cost_map(self, monkeypatch):
"""The 480p/1080p/4k tier keys resolve from the shipped runwayml cost map entries."""
from litellm.cost_calculator import completion_cost
local_map_path = os.path.join(
os.path.dirname(__file__), "..", "..", "model_prices_and_context_window.json"
)
with open(local_map_path, "r") as f:
monkeypatch.setattr(litellm, "model_cost", json.load(f))
def cost_for(model: str, resolution: str | None, duration: float) -> float:
mock_response = MagicMock()
mock_response.usage = {
"duration_seconds": duration,
**({"video_resolution": resolution} if resolution else {}),
}
type(mock_response)._hidden_params = {}
return completion_cost(
completion_response=mock_response,
model=model,
call_type="create_video",
custom_llm_provider="runwayml",
)
assert abs(cost_for("runwayml/seedance2", "4k", 8.0) - 12.0) < 0.001
assert abs(cost_for("runwayml/seedance2", "1080p", 8.0) - 3.2) < 0.001
assert abs(cost_for("runwayml/seedance2", "720p", 8.0) - 2.88) < 0.001
assert abs(cost_for("runwayml/seedance2_5", "480p", 8.0) - 1.6) < 0.001
assert abs(cost_for("runwayml/gen4.5", None, 8.0) - 0.96) < 0.001
def test_video_generation_with_files(self):
"""Test video generation with file uploads."""
config = OpenAIVideoConfig()
@ -2316,6 +2427,72 @@ def test_edit_and_extension_support_custom_provider_from_extra_body(
assert captured_data["custom_llm_provider"] == "vertex_ai"
@pytest.mark.parametrize(
"handler_name, path, form",
[
(
"video_edit",
"/v1/videos/edits",
{"model": "my-video-model", "prompt": "brighter", "video": "video_123"},
),
(
"video_extension",
"/v1/videos/extensions",
{"model": "my-video-model", "prompt": "continue", "seconds": "4", "video": "video_123"},
),
],
)
@pytest.mark.asyncio
async def test_edit_and_extension_read_cached_body_after_auth_consumes_stream(
handler_name, path, form
):
from urllib.parse import urlencode
from fastapi import Response
from starlette.requests import Request
import litellm.proxy.video_endpoints.endpoints as endpoints
from litellm.proxy._types import ProxyException, UserAPIKeyAuth
from litellm.proxy.common_utils.http_parsing_utils import _read_request_body
body = urlencode(form).encode()
stream = {"sent": False}
async def receive():
if stream["sent"]:
return {"type": "http.request", "body": b"", "more_body": False}
stream["sent"] = True
return {"type": "http.request", "body": body, "more_body": False}
request = Request(
{
"type": "http",
"method": "POST",
"path": path,
"headers": [
(b"content-type", b"application/x-www-form-urlencoded"),
(b"content-length", str(len(body)).encode()),
],
"query_string": b"",
},
receive,
)
await _read_request_body(request=request)
handler = getattr(endpoints, handler_name)
with pytest.raises(ProxyException) as exc_info:
await handler(
request=request,
fastapi_response=Response(),
user_api_key_dict=UserAPIKeyAuth(api_key="sk-1234"),
)
message = str(exc_info.value)
assert "Stream consumed" not in message
assert "my-video-model" in message
@pytest.mark.parametrize("endpoint", ["/v1/videos/edits", "/v1/videos/extensions"])
def test_edit_and_extension_route_with_encoded_video_ids(
video_proxy_test_client, endpoint

View file

@ -27624,6 +27624,10 @@ export interface components {
output_cost_per_second?: number | null;
/** Output Cost Per Second 1080P */
output_cost_per_second_1080p?: number | null;
/** Output Cost Per Second 480P */
output_cost_per_second_480p?: number | null;
/** Output Cost Per Second 4K */
output_cost_per_second_4k?: number | null;
/** Output Cost Per Token */
output_cost_per_token?: number | null;
/** Output Cost Per Token Above 128K Tokens */
@ -36833,6 +36837,10 @@ export interface components {
output_cost_per_second?: number | null;
/** Output Cost Per Second 1080P */
output_cost_per_second_1080p?: number | null;
/** Output Cost Per Second 480P */
output_cost_per_second_480p?: number | null;
/** Output Cost Per Second 4K */
output_cost_per_second_4k?: number | null;
/** Output Cost Per Token */
output_cost_per_token?: number | null;
/** Output Cost Per Token Above 128K Tokens */