mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
Merge remote-tracking branch 'origin/litellm_internal_staging' into litellm_harden_retry_breadcrumb_credentials
This commit is contained in:
commit
a1f1aa9cb5
40 changed files with 1307 additions and 208 deletions
|
|
@ -151,6 +151,7 @@ _VIDEO_CALL_TYPES: Final = frozenset(
|
|||
}
|
||||
)
|
||||
|
||||
|
||||
_SPEECH_CALL_TYPES: Final = frozenset(
|
||||
{
|
||||
CallTypes.speech.value,
|
||||
|
|
@ -1379,23 +1380,36 @@ def completion_cost(
|
|||
if custom_pricing and litellm_logging_obj is not None:
|
||||
_litellm_params = getattr(litellm_logging_obj, "litellm_params", None)
|
||||
if _litellm_params is not None:
|
||||
_metadata = _litellm_params.get("metadata", {}) or {}
|
||||
_video_model_info = _metadata.get("model_info", None)
|
||||
_video_model_info = next(
|
||||
(
|
||||
model_info
|
||||
for _metadata_key in ("metadata", "litellm_metadata")
|
||||
if (model_info := (_litellm_params.get(_metadata_key) or {}).get("model_info"))
|
||||
is not None
|
||||
),
|
||||
None,
|
||||
)
|
||||
|
||||
usage_obj = getattr(completion_response, "usage", None)
|
||||
duration_seconds: float | None = None
|
||||
video_resolution: str | None = None
|
||||
provider_reported_cost: float | None = None
|
||||
if completion_response is not None and usage_obj:
|
||||
# Handle both dict and Pydantic Usage object
|
||||
if isinstance(usage_obj, dict):
|
||||
duration_seconds = usage_obj.get("duration_seconds", None)
|
||||
_vr = usage_obj.get("video_resolution", None)
|
||||
provider_reported_cost = usage_obj.get("provider_reported_cost_usd", None)
|
||||
else:
|
||||
duration_seconds = getattr(usage_obj, "duration_seconds", None)
|
||||
_vr = getattr(usage_obj, "video_resolution", None)
|
||||
provider_reported_cost = getattr(usage_obj, "provider_reported_cost_usd", None)
|
||||
if _vr is not None:
|
||||
video_resolution = str(_vr).strip().lower()
|
||||
|
||||
if _video_model_info is None and provider_reported_cost is not None:
|
||||
return float(provider_reported_cost)
|
||||
|
||||
if duration_seconds is not None:
|
||||
# Calculate cost based on video duration using video-specific cost calculation
|
||||
from litellm.llms.openai.cost_calculation import (
|
||||
|
|
|
|||
|
|
@ -2301,6 +2301,7 @@ def exception_type(
|
|||
or custom_llm_provider == "custom_openai"
|
||||
or custom_llm_provider in litellm.openai_compatible_providers
|
||||
or custom_llm_provider == "mistral"
|
||||
or custom_llm_provider == "runwayml"
|
||||
):
|
||||
_map_openai_exception(
|
||||
model=model,
|
||||
|
|
|
|||
|
|
@ -440,6 +440,16 @@ class AnthropicModelInfo(BaseLLMModelInfo):
|
|||
"""
|
||||
return AnthropicModelInfo._supports_model_capability(model, "thinking_always_on", custom_llm_provider)
|
||||
|
||||
@staticmethod
|
||||
def _supports_legacy_thinking(model: str, custom_llm_provider: str) -> bool:
|
||||
"""Whether ``model`` is an adaptive-thinking model that still accepts legacy
|
||||
``thinking.type=enabled`` with ``budget_tokens`` (the Claude 4.6 family).
|
||||
The model cost map is authoritative: an explicit ``supports_legacy_thinking``
|
||||
entry resolved under ``custom_llm_provider``, or a ``fallback_generalizations``
|
||||
rule for unmapped 4.6 ids. Absent flag means the model rejects the legacy shape.
|
||||
"""
|
||||
return AnthropicModelInfo._supports_model_capability(model, "supports_legacy_thinking", custom_llm_provider)
|
||||
|
||||
@staticmethod
|
||||
def maybe_drop_disabled_thinking(
|
||||
model: str,
|
||||
|
|
|
|||
|
|
@ -379,13 +379,19 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
|
|||
def _translate_legacy_thinking_for_adaptive_model(
|
||||
model: str, optional_params: dict, custom_llm_provider: str
|
||||
) -> None:
|
||||
"""Translate legacy ``thinking.type=enabled`` to adaptive for 4.6/4.7.
|
||||
Caller-provided ``output_config.effort`` is never overridden.
|
||||
"""Translate legacy ``thinking.type=enabled`` to adaptive for the
|
||||
adaptive-thinking models that reject it (4.7+ and the 5 families).
|
||||
Models flagged ``supports_legacy_thinking`` (the 4.6 family) accept the
|
||||
legacy shape natively, so it is forwarded verbatim and the caller's
|
||||
``budget_tokens`` cap keeps applying. Caller-provided
|
||||
``output_config.effort`` is never overridden.
|
||||
"""
|
||||
from litellm.llms.anthropic.chat.transformation import AnthropicConfig
|
||||
|
||||
if not AnthropicModelInfo._is_adaptive_thinking_model(model, custom_llm_provider):
|
||||
return
|
||||
if AnthropicModelInfo._supports_legacy_thinking(model, custom_llm_provider):
|
||||
return
|
||||
thinking: Final = optional_params.get("thinking")
|
||||
if not isinstance(thinking, dict) or thinking.get("type") != "enabled":
|
||||
return
|
||||
|
|
|
|||
|
|
@ -134,14 +134,7 @@ def cost_per_second(model: str, custom_llm_provider: str | None, duration: float
|
|||
|
||||
|
||||
def _video_resolution_to_cost_field_suffix(resolution: str) -> str | None:
|
||||
"""
|
||||
Map usage resolution to a safe suffix for ``output_cost_per_second_<suffix>`` keys.
|
||||
|
||||
Note: Currently only ``output_cost_per_second_1080p`` is explicitly declared in
|
||||
ModelInfo (types/utils.py). Other resolution tiers (e.g., 720p, 4k) can be added
|
||||
to model_prices_and_context_window.json but are not exposed via get_model_info()
|
||||
until added to the ModelInfo TypedDict.
|
||||
"""
|
||||
"""Map usage resolution to a safe suffix for ``output_cost_per_second_<suffix>`` keys."""
|
||||
r: Final = resolution.strip().lower()
|
||||
if not r:
|
||||
return None
|
||||
|
|
|
|||
|
|
@ -1,5 +1,6 @@
|
|||
from collections.abc import Mapping, Sequence
|
||||
from datetime import datetime
|
||||
from types import MappingProxyType
|
||||
from typing import TYPE_CHECKING, Any, Final, Literal
|
||||
|
||||
import httpx
|
||||
|
|
@ -33,6 +34,10 @@ else:
|
|||
LiteLLMLoggingObj = Any
|
||||
|
||||
|
||||
class RunwayMLError(BaseLLMException):
|
||||
pass
|
||||
|
||||
|
||||
class _RunwayTaskResponse(TypedDict, total=False):
|
||||
id: ReadOnly[str]
|
||||
status: ReadOnly[str]
|
||||
|
|
@ -41,7 +46,8 @@ class _RunwayTaskResponse(TypedDict, total=False):
|
|||
output: ReadOnly[Sequence[str] | str]
|
||||
failureCode: ReadOnly[str]
|
||||
failure: ReadOnly[str]
|
||||
progress: ReadOnly[int]
|
||||
progress: ReadOnly[float]
|
||||
estimatedCost: ReadOnly[Mapping[str, float]]
|
||||
|
||||
|
||||
class _VideoObjectData(TypedDict, extra_items=object):
|
||||
|
|
@ -56,12 +62,54 @@ def _parse_runway_task_response(raw_response: httpx.Response) -> _RunwayTaskResp
|
|||
return response_data
|
||||
|
||||
|
||||
_USD_PER_CREDIT: Final = 0.01
|
||||
|
||||
_RESOLUTION_AREA_TIERS: Final[tuple[tuple[int, str], ...]] = (
|
||||
(600_000, "480p"),
|
||||
(1_500_000, "720p"),
|
||||
(4_000_000, "1080p"),
|
||||
)
|
||||
|
||||
|
||||
def _ratio_to_resolution(ratio: object) -> str | None:
|
||||
if not isinstance(ratio, str) or ":" not in ratio:
|
||||
return None
|
||||
width_str, _, height_str = ratio.partition(":")
|
||||
if not (width_str.isdigit() and height_str.isdigit()):
|
||||
return None
|
||||
area: Final = int(width_str) * int(height_str)
|
||||
return next((label for threshold, label in _RESOLUTION_AREA_TIERS if area < threshold), "4k")
|
||||
|
||||
|
||||
def _duration_seconds(seconds: str | None) -> float | None:
|
||||
if not seconds:
|
||||
return None
|
||||
try:
|
||||
return float(seconds)
|
||||
except ValueError:
|
||||
return None
|
||||
|
||||
|
||||
def _estimated_cost_usd(response_data: _RunwayTaskResponse) -> float | None:
|
||||
estimated_cost: Final = response_data.get("estimatedCost")
|
||||
if not isinstance(estimated_cost, Mapping):
|
||||
return None
|
||||
credits: Final = estimated_cost.get("credits")
|
||||
if not isinstance(credits, (int, float)):
|
||||
return None
|
||||
return float(credits) * _USD_PER_CREDIT
|
||||
|
||||
|
||||
def _progress_percent(progress: float) -> int:
|
||||
return min(100, max(0, round(float(progress) * 100)))
|
||||
|
||||
|
||||
class RunwayMLVideoConfig(BaseVideoConfig):
|
||||
"""
|
||||
Configuration class for RunwayML video generation.
|
||||
|
||||
RunwayML uses a task-based API where:
|
||||
1. POST /v1/image_to_video creates a task
|
||||
1. POST /v1/text_to_video, /v1/image_to_video, or /v1/video_to_video creates a task
|
||||
2. The task returns immediately with a task ID
|
||||
3. Client must poll or wait for task completion
|
||||
"""
|
||||
|
|
@ -195,31 +243,36 @@ class RunwayMLVideoConfig(BaseVideoConfig):
|
|||
"""
|
||||
Transform the video creation request for RunwayML API.
|
||||
|
||||
RunwayML expects:
|
||||
{
|
||||
"model": "gen4_turbo",
|
||||
"promptImage": "https://... or data:image/...",
|
||||
"promptText": "description",
|
||||
"ratio": "1280:720",
|
||||
"duration": 5
|
||||
}
|
||||
RunwayML has three generation endpoints discriminated by which input is
|
||||
present, and each request body rejects unknown fields:
|
||||
- /text_to_video: promptText only (rejects promptImage)
|
||||
- /image_to_video: promptImage (+ optional promptText)
|
||||
- /video_to_video: promptVideo or videoUri (rejects promptImage)
|
||||
"""
|
||||
# Build the request data
|
||||
merged_params: Final = MappingProxyType(
|
||||
{
|
||||
"model": model,
|
||||
"promptText": prompt,
|
||||
**video_create_optional_request_params,
|
||||
}
|
||||
)
|
||||
|
||||
endpoint: Final = self._select_generation_endpoint(merged_params)
|
||||
|
||||
request_data: Final[dict[str, object]] = {
|
||||
"model": model,
|
||||
"promptText": prompt,
|
||||
key: value for key, value in merged_params.items() if endpoint == "image_to_video" or key != "promptImage"
|
||||
}
|
||||
|
||||
# Add mapped parameters
|
||||
request_data.update(video_create_optional_request_params)
|
||||
|
||||
# RunwayML uses JSON body, no files multipart
|
||||
files_list: Final[RequestFiles] = []
|
||||
|
||||
# Append the specific endpoint for video generation
|
||||
full_api_base: Final = f"{api_base}/image_to_video"
|
||||
return request_data, files_list, f"{api_base}/{endpoint}"
|
||||
|
||||
return request_data, files_list, full_api_base
|
||||
def _select_generation_endpoint(self, request_data: Mapping[str, object]) -> str:
|
||||
if request_data.get("promptVideo") is not None or request_data.get("videoUri") is not None:
|
||||
return "video_to_video"
|
||||
if request_data.get("promptImage") is not None:
|
||||
return "image_to_video"
|
||||
return "text_to_video"
|
||||
|
||||
def transform_video_create_response(
|
||||
self,
|
||||
|
|
@ -285,13 +338,15 @@ class RunwayMLVideoConfig(BaseVideoConfig):
|
|||
video_obj.id = encode_video_id_with_provider(video_obj.id, custom_llm_provider, model)
|
||||
|
||||
# Add usage data for cost tracking
|
||||
usage_data: Final = {}
|
||||
if video_obj and hasattr(video_obj, "seconds") and video_obj.seconds:
|
||||
try:
|
||||
usage_data["duration_seconds"] = float(video_obj.seconds)
|
||||
except (ValueError, TypeError):
|
||||
pass
|
||||
video_obj.usage = usage_data
|
||||
video_obj.usage = {
|
||||
key: value
|
||||
for key, value in (
|
||||
("duration_seconds", _duration_seconds(video_obj.seconds)),
|
||||
("video_resolution", _ratio_to_resolution(request_data.get("ratio") if request_data else None)),
|
||||
("provider_reported_cost_usd", _estimated_cost_usd(response_data)),
|
||||
)
|
||||
if value is not None
|
||||
}
|
||||
|
||||
return video_obj
|
||||
|
||||
|
|
@ -581,8 +636,9 @@ class RunwayMLVideoConfig(BaseVideoConfig):
|
|||
if "completedAt" in response_data:
|
||||
video_data["completed_at"] = self._parse_runway_timestamp(response_data.get("completedAt"))
|
||||
|
||||
if "progress" in response_data:
|
||||
video_data["progress"] = response_data["progress"]
|
||||
progress_value: Final = response_data.get("progress")
|
||||
if progress_value is not None:
|
||||
video_data["progress"] = _progress_percent(progress_value)
|
||||
|
||||
if "failureCode" in response_data or "failure" in response_data:
|
||||
video_data["error"] = {
|
||||
|
|
@ -646,9 +702,7 @@ class RunwayMLVideoConfig(BaseVideoConfig):
|
|||
raise NotImplementedError("video extension is not supported for RunwayML")
|
||||
|
||||
def get_error_class(self, error_message: str, status_code: int, headers: dict | httpx.Headers) -> BaseLLMException:
|
||||
from ...base_llm.chat.transformation import BaseLLMException
|
||||
|
||||
raise BaseLLMException(
|
||||
return RunwayMLError(
|
||||
status_code=status_code,
|
||||
message=error_message,
|
||||
headers=headers,
|
||||
|
|
|
|||
|
|
@ -1019,6 +1019,7 @@
|
|||
},
|
||||
"anthropic.claude-opus-4-6-v1": {
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 1e-05,
|
||||
"cache_read_input_token_cost": 5e-07,
|
||||
|
|
@ -1053,6 +1054,7 @@
|
|||
},
|
||||
"global.anthropic.claude-opus-4-6-v1": {
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 1e-05,
|
||||
"cache_read_input_token_cost": 5e-07,
|
||||
|
|
@ -1087,6 +1089,7 @@
|
|||
},
|
||||
"us.anthropic.claude-opus-4-6-v1": {
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 6.875e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 1.1e-05,
|
||||
"cache_read_input_token_cost": 5.5e-07,
|
||||
|
|
@ -1121,6 +1124,7 @@
|
|||
},
|
||||
"eu.anthropic.claude-opus-4-6-v1": {
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 6.875e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 1.1e-05,
|
||||
"cache_read_input_token_cost": 5.5e-07,
|
||||
|
|
@ -1155,6 +1159,7 @@
|
|||
},
|
||||
"au.anthropic.claude-opus-4-6-v1": {
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 6.875e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 1.1e-05,
|
||||
"cache_read_input_token_cost": 5.5e-07,
|
||||
|
|
@ -2233,6 +2238,7 @@
|
|||
},
|
||||
"anthropic.claude-sonnet-4-6": {
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 3.75e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 6e-06,
|
||||
"cache_read_input_token_cost": 3e-07,
|
||||
|
|
@ -2266,6 +2272,7 @@
|
|||
},
|
||||
"global.anthropic.claude-sonnet-4-6": {
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 3.75e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 6e-06,
|
||||
"cache_read_input_token_cost": 3e-07,
|
||||
|
|
@ -2299,6 +2306,7 @@
|
|||
},
|
||||
"us.anthropic.claude-sonnet-4-6": {
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 4.125e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 6.6e-06,
|
||||
"cache_read_input_token_cost": 3.3e-07,
|
||||
|
|
@ -2332,6 +2340,7 @@
|
|||
},
|
||||
"eu.anthropic.claude-sonnet-4-6": {
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 4.125e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 6.6e-06,
|
||||
"cache_read_input_token_cost": 3.3e-07,
|
||||
|
|
@ -2365,6 +2374,7 @@
|
|||
},
|
||||
"au.anthropic.claude-sonnet-4-6": {
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 4.125e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 6.6e-06,
|
||||
"cache_read_input_token_cost": 3.3e-07,
|
||||
|
|
@ -2398,6 +2408,7 @@
|
|||
},
|
||||
"jp.anthropic.claude-sonnet-4-6": {
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 4.125e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 6.6e-06,
|
||||
"cache_read_input_token_cost": 3.3e-07,
|
||||
|
|
@ -2950,6 +2961,7 @@
|
|||
"azure_ai/claude-opus-4-6": {
|
||||
"deprecation_date": "2027-02-02",
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"input_cost_per_token": 5e-06,
|
||||
"output_cost_per_token": 2.5e-05,
|
||||
"litellm_provider": "azure_ai",
|
||||
|
|
@ -3181,6 +3193,7 @@
|
|||
"azure_ai/claude-sonnet-4-6": {
|
||||
"deprecation_date": "2027-02-10",
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 3.75e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 6e-06,
|
||||
"cache_read_input_token_cost": 3e-07,
|
||||
|
|
@ -12489,6 +12502,7 @@
|
|||
"search_context_size_medium": 0.01
|
||||
},
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_computer_use": true,
|
||||
"supports_function_calling": true,
|
||||
|
|
@ -12698,6 +12712,7 @@
|
|||
"search_context_size_medium": 0.01
|
||||
},
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"supports_assistant_prefill": false,
|
||||
"supports_computer_use": true,
|
||||
"supports_function_calling": true,
|
||||
|
|
@ -12735,6 +12750,7 @@
|
|||
"search_context_size_medium": 0.01
|
||||
},
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"supports_assistant_prefill": false,
|
||||
"supports_computer_use": true,
|
||||
"supports_function_calling": true,
|
||||
|
|
@ -14724,6 +14740,7 @@
|
|||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true
|
||||
|
|
@ -14892,6 +14909,7 @@
|
|||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true
|
||||
|
|
@ -23427,6 +23445,7 @@
|
|||
},
|
||||
"github_copilot/claude-opus-4.6-fast": {
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"litellm_provider": "github_copilot",
|
||||
"max_input_tokens": 128000,
|
||||
"max_output_tokens": 16000,
|
||||
|
|
@ -33810,6 +33829,7 @@
|
|||
},
|
||||
"openrouter/anthropic/claude-sonnet-4.6": {
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 3.75e-06,
|
||||
"cache_creation_input_token_cost_above_200k_tokens": 7.5e-06,
|
||||
"cache_read_input_token_cost": 3e-07,
|
||||
|
|
@ -33854,6 +33874,7 @@
|
|||
},
|
||||
"openrouter/anthropic/claude-opus-4.6": {
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
"cache_read_input_token_cost": 5e-07,
|
||||
"input_cost_per_token": 5e-06,
|
||||
|
|
@ -35928,6 +35949,7 @@
|
|||
},
|
||||
"perplexity/anthropic/claude-opus-4-6": {
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"litellm_provider": "perplexity",
|
||||
"mode": "responses",
|
||||
"supports_web_search": true,
|
||||
|
|
@ -39303,6 +39325,7 @@
|
|||
},
|
||||
"vercel_ai_gateway/anthropic/claude-opus-4.6": {
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
"cache_read_input_token_cost": 5e-07,
|
||||
"input_cost_per_token": 5e-06,
|
||||
|
|
@ -40562,6 +40585,7 @@
|
|||
"deprecation_date": "2027-02-05",
|
||||
"regional_endpoint_uplift_multiplier": 1.1,
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 1e-05,
|
||||
"cache_read_input_token_cost": 5e-07,
|
||||
|
|
@ -40594,6 +40618,7 @@
|
|||
"deprecation_date": "2027-02-05",
|
||||
"regional_endpoint_uplift_multiplier": 1.1,
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 1e-05,
|
||||
"cache_read_input_token_cost": 5e-07,
|
||||
|
|
@ -40959,6 +40984,7 @@
|
|||
"vertex_ai/claude-sonnet-4-6": {
|
||||
"regional_endpoint_uplift_multiplier": 1.1,
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 3.75e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 6e-06,
|
||||
"cache_read_input_token_cost": 3e-07,
|
||||
|
|
@ -44042,10 +44068,10 @@
|
|||
"comment": "5 credits per second @ $0.01 per credit = $0.05 per second"
|
||||
}
|
||||
},
|
||||
"runwayml/gen4_aleph": {
|
||||
"runwayml/gen4.5": {
|
||||
"litellm_provider": "runwayml",
|
||||
"mode": "video_generation",
|
||||
"output_cost_per_video_per_second": 0.15,
|
||||
"output_cost_per_second": 0.12,
|
||||
"source": "https://docs.dev.runwayml.com/guides/pricing/",
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
|
|
@ -44055,13 +44081,136 @@
|
|||
"video"
|
||||
],
|
||||
"metadata": {
|
||||
"comment": "15 credits per second @ $0.01 per credit = $0.15 per second"
|
||||
"comment": "12 credits per second @ $0.01 per credit = $0.12 per second"
|
||||
}
|
||||
},
|
||||
"runwayml/gen3a_turbo": {
|
||||
"runwayml/aleph2": {
|
||||
"litellm_provider": "runwayml",
|
||||
"mode": "video_generation",
|
||||
"output_cost_per_video_per_second": 0.05,
|
||||
"output_cost_per_second": 0.28,
|
||||
"source": "https://docs.dev.runwayml.com/guides/pricing/",
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"video"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"video"
|
||||
],
|
||||
"metadata": {
|
||||
"comment": "28 credits per second @ $0.01 per credit = $0.28 per second; 56 credit minimum per task not modeled"
|
||||
}
|
||||
},
|
||||
"runwayml/seedance2": {
|
||||
"litellm_provider": "runwayml",
|
||||
"mode": "video_generation",
|
||||
"output_cost_per_second": 0.36,
|
||||
"output_cost_per_second_1080p": 0.4,
|
||||
"output_cost_per_second_4k": 1.5,
|
||||
"source": "https://docs.dev.runwayml.com/guides/pricing/",
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image",
|
||||
"video"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"video"
|
||||
],
|
||||
"metadata": {
|
||||
"comment": "36 credits per second at 480p/720p, 40 at 1080p, 150 at 4K @ $0.01 per credit"
|
||||
}
|
||||
},
|
||||
"runwayml/seedance2_fast": {
|
||||
"litellm_provider": "runwayml",
|
||||
"mode": "video_generation",
|
||||
"output_cost_per_second": 0.29,
|
||||
"source": "https://docs.dev.runwayml.com/guides/pricing/",
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image",
|
||||
"video"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"video"
|
||||
],
|
||||
"metadata": {
|
||||
"comment": "29 credits per second at 480p/720p @ $0.01 per credit = $0.29 per second"
|
||||
}
|
||||
},
|
||||
"runwayml/seedance2_mini": {
|
||||
"litellm_provider": "runwayml",
|
||||
"mode": "video_generation",
|
||||
"output_cost_per_second": 0.16,
|
||||
"source": "https://docs.dev.runwayml.com/guides/pricing/",
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image",
|
||||
"video"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"video"
|
||||
],
|
||||
"metadata": {
|
||||
"comment": "16 credits per second @ $0.01 per credit = $0.16 per second; 64 credit minimum per task not modeled"
|
||||
}
|
||||
},
|
||||
"runwayml/seedance2_5": {
|
||||
"litellm_provider": "runwayml",
|
||||
"mode": "video_generation",
|
||||
"output_cost_per_second": 0.3,
|
||||
"output_cost_per_second_480p": 0.2,
|
||||
"output_cost_per_second_1080p": 0.68,
|
||||
"source": "https://docs.dev.runwayml.com/guides/pricing/",
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image",
|
||||
"video"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"video"
|
||||
],
|
||||
"metadata": {
|
||||
"comment": "Output: 20/30/68 credits per second at 480p/720p/1080p @ $0.01 per credit; input video billed additionally at 10/15/34 credits per input second and the 80 credit minimum per task are not modeled"
|
||||
}
|
||||
},
|
||||
"runwayml/hailuo3": {
|
||||
"litellm_provider": "runwayml",
|
||||
"mode": "video_generation",
|
||||
"output_cost_per_second": 0.1,
|
||||
"output_cost_per_second_1080p": 0.15,
|
||||
"source": "https://docs.dev.runwayml.com/guides/pricing/",
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image",
|
||||
"video"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"video"
|
||||
],
|
||||
"metadata": {
|
||||
"comment": "10 credits per second at 768P, 15 at 2K (mapped to the 1080p tier) @ $0.01 per credit; 2 credits per reference image not modeled"
|
||||
}
|
||||
},
|
||||
"runwayml/gemini_omni_flash": {
|
||||
"litellm_provider": "runwayml",
|
||||
"mode": "video_generation",
|
||||
"output_cost_per_second": 0.1,
|
||||
"source": "https://docs.dev.runwayml.com/guides/pricing/",
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image",
|
||||
"video"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"video"
|
||||
],
|
||||
"metadata": {
|
||||
"comment": "10 credits per second @ $0.01 per credit = $0.10 per second"
|
||||
}
|
||||
},
|
||||
"runwayml/veo3.1": {
|
||||
"litellm_provider": "runwayml",
|
||||
"mode": "video_generation",
|
||||
"output_cost_per_second": 0.4,
|
||||
"source": "https://docs.dev.runwayml.com/guides/pricing/",
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
|
|
@ -44071,7 +44220,23 @@
|
|||
"video"
|
||||
],
|
||||
"metadata": {
|
||||
"comment": "5 credits per second @ $0.01 per credit = $0.05 per second"
|
||||
"comment": "40 credits per second with audio, 20 without @ $0.01 per credit; priced at the with-audio rate"
|
||||
}
|
||||
},
|
||||
"runwayml/veo3.1_fast": {
|
||||
"litellm_provider": "runwayml",
|
||||
"mode": "video_generation",
|
||||
"output_cost_per_second": 0.15,
|
||||
"source": "https://docs.dev.runwayml.com/guides/pricing/",
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"video"
|
||||
],
|
||||
"metadata": {
|
||||
"comment": "15 credits per second with audio, 10 without @ $0.01 per credit; priced at the with-audio rate"
|
||||
}
|
||||
},
|
||||
"runwayml/gen4_image": {
|
||||
|
|
@ -48728,6 +48893,7 @@
|
|||
"vertex_ai/claude-sonnet-4-6@default": {
|
||||
"regional_endpoint_uplift_multiplier": 1.1,
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 3.75e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 6e-06,
|
||||
"cache_read_input_token_cost": 3e-07,
|
||||
|
|
@ -49512,6 +49678,7 @@
|
|||
},
|
||||
"snowflake/claude-sonnet-4-6": {
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"max_tokens": 16384,
|
||||
"max_input_tokens": 200000,
|
||||
"max_output_tokens": 16384,
|
||||
|
|
@ -50502,6 +50669,14 @@
|
|||
"supports_adaptive_thinking": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "claude-legacy-thinking",
|
||||
"pattern": "claude-[a-z]+-4[-._]6(?!\\d)",
|
||||
"description": "Claude at version 4.6 exactly, in any id shape that contains claude-<family>-4-6 (dotted and underscored minors included, dated releases such as claude-sonnet-4-6-20260219 too). The 4.6 family is adaptive-thinking yet still accepts legacy thinking.type=enabled with budget_tokens, so the caller's hard budget cap is forwarded verbatim instead of being rewritten to an uncapped output_config.effort. The lookahead keeps two-digit minors such as 4-60 from matching. 4.7+ and 5+ majors reject the legacy shape and stay on the adaptive translation.",
|
||||
"model_info": {
|
||||
"supports_legacy_thinking": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "claude-always-on-thinking",
|
||||
"pattern": "claude-(?:fable|mythos)-",
|
||||
|
|
|
|||
|
|
@ -274,10 +274,8 @@ async def get_form_data(request: Request) -> dict[str, Any]:
|
|||
Handles when OpenAI SDKs pass form keys as `timestamp_granularities[]="word"` instead of `timestamp_granularities=["word", "sentence"]`
|
||||
"""
|
||||
form: Final = await request.form()
|
||||
form_data: Final = dict(form)
|
||||
parsed_form_data: Final[dict[str, Any]] = {}
|
||||
for key, value in form_data.items():
|
||||
# OpenAI SDKs pass form keys as `timestamp_granularities[]="word"` instead of `timestamp_granularities=["word", "sentence"]`
|
||||
for key, value in form.multi_items(): # not dict(form), which keeps only the last repeat
|
||||
if key.endswith("[]"):
|
||||
clean_key = key[:-2]
|
||||
parsed_form_data.setdefault(clean_key, []).append(value)
|
||||
|
|
|
|||
|
|
@ -433,6 +433,9 @@ def _update_litellm_params_for_health_check(model_info: dict, litellm_params: di
|
|||
"""
|
||||
Update the litellm params for health check.
|
||||
|
||||
- merges `model_info.health_check_params` into the probe request, so a deployment whose provider
|
||||
requires a payload field litellm does not synthesize (e.g. `mediaSource` for Bedrock TwelveLabs
|
||||
Pegasus) can supply it. The dedicated knobs below are applied afterwards and win on conflict.
|
||||
- gets a short `messages` param for health check
|
||||
- adds a bounded `max_tokens` when the deployment is a chat-style mode
|
||||
(`chat`, `completion`, `responses`) or the operator explicitly opts in
|
||||
|
|
@ -447,6 +450,16 @@ def _update_litellm_params_for_health_check(model_info: dict, litellm_params: di
|
|||
model_info,
|
||||
litellm_params, # any-ok: untyped router config dict
|
||||
)
|
||||
_health_check_params: Final = model_info.get("health_check_params", None)
|
||||
if isinstance(_health_check_params, dict):
|
||||
litellm_params.update(_health_check_params)
|
||||
elif _health_check_params is not None:
|
||||
logger.warning(
|
||||
"health_check_params for model %s is a %s, expected a dict. Ignoring it.",
|
||||
litellm_params.get("model"),
|
||||
type(_health_check_params).__name__,
|
||||
)
|
||||
|
||||
litellm_params["messages"] = _get_random_llm_message()
|
||||
if _should_inject_health_check_max_tokens(
|
||||
model_info,
|
||||
|
|
|
|||
|
|
@ -1888,6 +1888,8 @@ async def test_model_connection(
|
|||
# already resolved before reaching this endpoint; any remaining
|
||||
# reference must have come from the request body.
|
||||
_reject_os_environ_references(request_litellm_params)
|
||||
if model_info:
|
||||
_reject_os_environ_references(model_info)
|
||||
model_name: Final = request_litellm_params.get("model")
|
||||
|
||||
# Look up model configuration from router if model name is provided
|
||||
|
|
@ -1950,23 +1952,23 @@ async def test_model_connection(
|
|||
**request_litellm_params,
|
||||
}
|
||||
|
||||
## Auth check
|
||||
auth_model_info: Final = loaded_model_info if loaded_model_info is not None else model_info
|
||||
resolved_model_info: Final = loaded_model_info if loaded_model_info is not None else model_info
|
||||
litellm_params = _update_litellm_params_for_health_check(
|
||||
model_info=resolved_model_info or {},
|
||||
litellm_params=litellm_params,
|
||||
)
|
||||
|
||||
## Auth check, on the final probe params so health_check_params cannot retarget it afterwards
|
||||
await ModelManagementAuthChecks.can_user_make_model_call(
|
||||
model_params=Deployment(
|
||||
model_name="test_model",
|
||||
litellm_params=LiteLLM_Params(**litellm_params),
|
||||
model_info=auth_model_info,
|
||||
model_info=resolved_model_info,
|
||||
),
|
||||
user_api_key_dict=user_api_key_dict,
|
||||
prisma_client=prisma_client,
|
||||
premium_user=premium_user,
|
||||
)
|
||||
# Include health_check_params if provided
|
||||
litellm_params = _update_litellm_params_for_health_check(
|
||||
model_info={},
|
||||
litellm_params=litellm_params,
|
||||
)
|
||||
mode = mode or litellm_params.pop("mode", None)
|
||||
|
||||
result: Final = await run_with_timeout(
|
||||
|
|
|
|||
|
|
@ -2,7 +2,6 @@
|
|||
|
||||
from typing import Any, Final
|
||||
|
||||
import orjson
|
||||
from fastapi import APIRouter, Depends, File, Form, Request, Response, UploadFile
|
||||
from fastapi.responses import ORJSONResponse
|
||||
|
||||
|
|
@ -20,6 +19,7 @@ from litellm.proxy.video_endpoints.utils import (
|
|||
encode_character_id_in_response,
|
||||
extract_model_from_target_model_names,
|
||||
get_custom_provider_from_data,
|
||||
video_reference_to_id,
|
||||
)
|
||||
from litellm.types.videos.utils import (
|
||||
decode_character_id_with_provider,
|
||||
|
|
@ -451,9 +451,7 @@ async def video_remix(
|
|||
version,
|
||||
)
|
||||
|
||||
# Read request body
|
||||
body: Final = await request.body()
|
||||
data: Final = orjson.loads(body)
|
||||
data: Final = await _read_request_body(request=request)
|
||||
data["video_id"] = video_id
|
||||
|
||||
decoded: Final = decode_video_id_with_provider(video_id)
|
||||
|
|
@ -760,15 +758,10 @@ async def video_edit(
|
|||
version,
|
||||
)
|
||||
|
||||
body: Final = await request.body()
|
||||
data: Final = orjson.loads(body)
|
||||
data: Final = await _read_request_body(request=request)
|
||||
data["video_id"] = video_reference_to_id(data.pop("video", None))
|
||||
|
||||
# Extract video_id from nested video object
|
||||
video_ref: Final = data.pop("video", {})
|
||||
video_id: Final = video_ref.get("id", "") if isinstance(video_ref, dict) else ""
|
||||
data["video_id"] = video_id
|
||||
|
||||
decoded: Final = decode_video_id_with_provider(video_id)
|
||||
decoded: Final = decode_video_id_with_provider(data["video_id"])
|
||||
provider_from_id: Final = decoded.get("custom_llm_provider")
|
||||
model_id_from_decoded: Final = decoded.get("model_id")
|
||||
|
||||
|
|
@ -860,15 +853,10 @@ async def video_extension(
|
|||
version,
|
||||
)
|
||||
|
||||
body: Final = await request.body()
|
||||
data: Final = orjson.loads(body)
|
||||
data: Final = await _read_request_body(request=request)
|
||||
data["video_id"] = video_reference_to_id(data.pop("video", None))
|
||||
|
||||
# Extract video_id from nested video object
|
||||
video_ref: Final = data.pop("video", {})
|
||||
video_id: Final = video_ref.get("id", "") if isinstance(video_ref, dict) else ""
|
||||
data["video_id"] = video_id
|
||||
|
||||
decoded: Final = decode_video_id_with_provider(video_id)
|
||||
decoded: Final = decode_video_id_with_provider(data["video_id"])
|
||||
provider_from_id: Final = decoded.get("custom_llm_provider")
|
||||
model_id_from_decoded: Final = decoded.get("model_id")
|
||||
|
||||
|
|
|
|||
|
|
@ -13,6 +13,18 @@ def extract_model_from_target_model_names(target_model_names: Any) -> str | None
|
|||
return target_model_names[0] if target_model_names else None
|
||||
|
||||
|
||||
def video_reference_to_id(video_ref: object) -> str:
|
||||
if isinstance(video_ref, dict):
|
||||
return video_ref.get("id", "")
|
||||
if not isinstance(video_ref, str):
|
||||
return ""
|
||||
try:
|
||||
parsed_ref: Final = orjson.loads(video_ref)
|
||||
except orjson.JSONDecodeError:
|
||||
return video_ref
|
||||
return parsed_ref.get("id", "") if isinstance(parsed_ref, dict) else video_ref
|
||||
|
||||
|
||||
def get_custom_provider_from_data(data: dict[str, Any]) -> str | None:
|
||||
custom_llm_provider: Final = data.get("custom_llm_provider")
|
||||
if custom_llm_provider:
|
||||
|
|
|
|||
|
|
@ -10,7 +10,7 @@ from typing import Any, ClassVar, Final, Generic, Literal, TypeVar, get_type_hin
|
|||
|
||||
import httpx
|
||||
from pydantic import BaseModel, ConfigDict, Field, field_validator, model_validator
|
||||
from typing_extensions import Protocol, Required, TypedDict, runtime_checkable
|
||||
from typing_extensions import Protocol, ReadOnly, Required, TypedDict, runtime_checkable
|
||||
|
||||
from litellm._uuid import uuid
|
||||
|
||||
|
|
@ -480,7 +480,9 @@ class LiteLLMParamsTypedDict(TypedDict, total=False):
|
|||
output_cost_per_token: float | None
|
||||
input_cost_per_second: float | None
|
||||
output_cost_per_second: float | None
|
||||
output_cost_per_second_480p: ReadOnly[float | None]
|
||||
output_cost_per_second_1080p: float | None
|
||||
output_cost_per_second_4k: ReadOnly[float | None]
|
||||
num_retries: int | None
|
||||
## MOCK RESPONSES ##
|
||||
mock_response: str | ModelResponse | Exception | None
|
||||
|
|
|
|||
|
|
@ -154,6 +154,7 @@ class ProviderSpecificModelInfo(TypedDict, total=False):
|
|||
supports_web_search: bool | None
|
||||
supports_reasoning: bool | None
|
||||
supports_adaptive_thinking: bool | None
|
||||
supports_legacy_thinking: ReadOnly[bool | None]
|
||||
thinking_always_on: ReadOnly[bool | None]
|
||||
supports_tool_search: bool | None
|
||||
supports_mid_conversation_system: bool | None
|
||||
|
|
@ -277,6 +278,8 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False):
|
|||
output_cost_per_second_1080p: (
|
||||
float | None
|
||||
) # video_generation tier: key output_cost_per_second_<resolution> (e.g. 1080p, 720p)
|
||||
output_cost_per_second_480p: ReadOnly[float | None]
|
||||
output_cost_per_second_4k: ReadOnly[float | None]
|
||||
ocr_cost_per_page: float | None # for OCR models
|
||||
ocr_cost_per_credit: float | None # for OCR models priced by credit
|
||||
annotation_cost_per_page: float | None # for OCR models
|
||||
|
|
@ -3337,6 +3340,8 @@ class CustomPricingLiteLLMParams(MirroredPricingParams):
|
|||
input_cost_per_second: float | None = None
|
||||
output_cost_per_second: float | None = None
|
||||
output_cost_per_second_1080p: float | None = None
|
||||
output_cost_per_second_480p: float | None = None
|
||||
output_cost_per_second_4k: float | None = None
|
||||
input_cost_per_pixel: float | None = None
|
||||
output_cost_per_pixel: float | None = None
|
||||
|
||||
|
|
|
|||
|
|
@ -5726,6 +5726,8 @@ def _get_model_info_helper(
|
|||
),
|
||||
output_cost_per_second=_model_info.get("output_cost_per_second", None),
|
||||
output_cost_per_second_1080p=_model_info.get("output_cost_per_second_1080p", None),
|
||||
output_cost_per_second_480p=_model_info.get("output_cost_per_second_480p", None),
|
||||
output_cost_per_second_4k=_model_info.get("output_cost_per_second_4k", None),
|
||||
output_cost_per_video_per_second=_model_info.get("output_cost_per_video_per_second", None),
|
||||
output_cost_per_image=_model_info.get("output_cost_per_image", None),
|
||||
output_cost_per_image_token=_model_info.get("output_cost_per_image_token", None),
|
||||
|
|
@ -5753,6 +5755,7 @@ def _get_model_info_helper(
|
|||
supports_url_context=_model_info.get("supports_url_context", None),
|
||||
supports_reasoning=_model_info.get("supports_reasoning", None),
|
||||
supports_adaptive_thinking=_model_info.get("supports_adaptive_thinking", None),
|
||||
supports_legacy_thinking=_model_info.get("supports_legacy_thinking", None),
|
||||
thinking_always_on=_model_info.get("thinking_always_on", None),
|
||||
supports_tool_search=_model_info.get("supports_tool_search", None),
|
||||
supports_mid_conversation_system=_model_info.get("supports_mid_conversation_system", None),
|
||||
|
|
@ -6526,21 +6529,10 @@ def acreate(*args, **kwargs): ## Thin client to handle the acreate langchain ca
|
|||
|
||||
|
||||
def prompt_token_calculator(model, messages):
|
||||
# use tiktoken or anthropic's tokenizer depending on the model
|
||||
text: Final = " ".join(message["content"] for message in messages)
|
||||
num_tokens = 0
|
||||
if "claude" in model:
|
||||
try:
|
||||
import anthropic
|
||||
except Exception:
|
||||
Exception("Anthropic import failed please run `pip install anthropic`")
|
||||
from anthropic import AI_PROMPT, HUMAN_PROMPT, Anthropic
|
||||
|
||||
anthropic_obj: Final = Anthropic()
|
||||
num_tokens = anthropic_obj.count_tokens(text)
|
||||
else:
|
||||
num_tokens = len(_get_default_encoding().encode(text))
|
||||
return num_tokens
|
||||
return token_counter(model=model, text=text)
|
||||
return len(_get_default_encoding().encode(text))
|
||||
|
||||
|
||||
def valid_model(model):
|
||||
|
|
|
|||
|
|
@ -1019,6 +1019,7 @@
|
|||
},
|
||||
"anthropic.claude-opus-4-6-v1": {
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 1e-05,
|
||||
"cache_read_input_token_cost": 5e-07,
|
||||
|
|
@ -1053,6 +1054,7 @@
|
|||
},
|
||||
"global.anthropic.claude-opus-4-6-v1": {
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 1e-05,
|
||||
"cache_read_input_token_cost": 5e-07,
|
||||
|
|
@ -1087,6 +1089,7 @@
|
|||
},
|
||||
"us.anthropic.claude-opus-4-6-v1": {
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 6.875e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 1.1e-05,
|
||||
"cache_read_input_token_cost": 5.5e-07,
|
||||
|
|
@ -1121,6 +1124,7 @@
|
|||
},
|
||||
"eu.anthropic.claude-opus-4-6-v1": {
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 6.875e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 1.1e-05,
|
||||
"cache_read_input_token_cost": 5.5e-07,
|
||||
|
|
@ -1155,6 +1159,7 @@
|
|||
},
|
||||
"au.anthropic.claude-opus-4-6-v1": {
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 6.875e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 1.1e-05,
|
||||
"cache_read_input_token_cost": 5.5e-07,
|
||||
|
|
@ -2233,6 +2238,7 @@
|
|||
},
|
||||
"anthropic.claude-sonnet-4-6": {
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 3.75e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 6e-06,
|
||||
"cache_read_input_token_cost": 3e-07,
|
||||
|
|
@ -2266,6 +2272,7 @@
|
|||
},
|
||||
"global.anthropic.claude-sonnet-4-6": {
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 3.75e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 6e-06,
|
||||
"cache_read_input_token_cost": 3e-07,
|
||||
|
|
@ -2299,6 +2306,7 @@
|
|||
},
|
||||
"us.anthropic.claude-sonnet-4-6": {
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 4.125e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 6.6e-06,
|
||||
"cache_read_input_token_cost": 3.3e-07,
|
||||
|
|
@ -2332,6 +2340,7 @@
|
|||
},
|
||||
"eu.anthropic.claude-sonnet-4-6": {
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 4.125e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 6.6e-06,
|
||||
"cache_read_input_token_cost": 3.3e-07,
|
||||
|
|
@ -2365,6 +2374,7 @@
|
|||
},
|
||||
"au.anthropic.claude-sonnet-4-6": {
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 4.125e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 6.6e-06,
|
||||
"cache_read_input_token_cost": 3.3e-07,
|
||||
|
|
@ -2398,6 +2408,7 @@
|
|||
},
|
||||
"jp.anthropic.claude-sonnet-4-6": {
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 4.125e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 6.6e-06,
|
||||
"cache_read_input_token_cost": 3.3e-07,
|
||||
|
|
@ -2950,6 +2961,7 @@
|
|||
"azure_ai/claude-opus-4-6": {
|
||||
"deprecation_date": "2027-02-02",
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"input_cost_per_token": 5e-06,
|
||||
"output_cost_per_token": 2.5e-05,
|
||||
"litellm_provider": "azure_ai",
|
||||
|
|
@ -3181,6 +3193,7 @@
|
|||
"azure_ai/claude-sonnet-4-6": {
|
||||
"deprecation_date": "2027-02-10",
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 3.75e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 6e-06,
|
||||
"cache_read_input_token_cost": 3e-07,
|
||||
|
|
@ -12489,6 +12502,7 @@
|
|||
"search_context_size_medium": 0.01
|
||||
},
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_computer_use": true,
|
||||
"supports_function_calling": true,
|
||||
|
|
@ -12698,6 +12712,7 @@
|
|||
"search_context_size_medium": 0.01
|
||||
},
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"supports_assistant_prefill": false,
|
||||
"supports_computer_use": true,
|
||||
"supports_function_calling": true,
|
||||
|
|
@ -12735,6 +12750,7 @@
|
|||
"search_context_size_medium": 0.01
|
||||
},
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"supports_assistant_prefill": false,
|
||||
"supports_computer_use": true,
|
||||
"supports_function_calling": true,
|
||||
|
|
@ -14724,6 +14740,7 @@
|
|||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true
|
||||
|
|
@ -14892,6 +14909,7 @@
|
|||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true
|
||||
|
|
@ -23427,6 +23445,7 @@
|
|||
},
|
||||
"github_copilot/claude-opus-4.6-fast": {
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"litellm_provider": "github_copilot",
|
||||
"max_input_tokens": 128000,
|
||||
"max_output_tokens": 16000,
|
||||
|
|
@ -33810,6 +33829,7 @@
|
|||
},
|
||||
"openrouter/anthropic/claude-sonnet-4.6": {
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 3.75e-06,
|
||||
"cache_creation_input_token_cost_above_200k_tokens": 7.5e-06,
|
||||
"cache_read_input_token_cost": 3e-07,
|
||||
|
|
@ -33854,6 +33874,7 @@
|
|||
},
|
||||
"openrouter/anthropic/claude-opus-4.6": {
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
"cache_read_input_token_cost": 5e-07,
|
||||
"input_cost_per_token": 5e-06,
|
||||
|
|
@ -35928,6 +35949,7 @@
|
|||
},
|
||||
"perplexity/anthropic/claude-opus-4-6": {
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"litellm_provider": "perplexity",
|
||||
"mode": "responses",
|
||||
"supports_web_search": true,
|
||||
|
|
@ -39303,6 +39325,7 @@
|
|||
},
|
||||
"vercel_ai_gateway/anthropic/claude-opus-4.6": {
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
"cache_read_input_token_cost": 5e-07,
|
||||
"input_cost_per_token": 5e-06,
|
||||
|
|
@ -40562,6 +40585,7 @@
|
|||
"deprecation_date": "2027-02-05",
|
||||
"regional_endpoint_uplift_multiplier": 1.1,
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 1e-05,
|
||||
"cache_read_input_token_cost": 5e-07,
|
||||
|
|
@ -40594,6 +40618,7 @@
|
|||
"deprecation_date": "2027-02-05",
|
||||
"regional_endpoint_uplift_multiplier": 1.1,
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 1e-05,
|
||||
"cache_read_input_token_cost": 5e-07,
|
||||
|
|
@ -40959,6 +40984,7 @@
|
|||
"vertex_ai/claude-sonnet-4-6": {
|
||||
"regional_endpoint_uplift_multiplier": 1.1,
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 3.75e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 6e-06,
|
||||
"cache_read_input_token_cost": 3e-07,
|
||||
|
|
@ -44042,10 +44068,10 @@
|
|||
"comment": "5 credits per second @ $0.01 per credit = $0.05 per second"
|
||||
}
|
||||
},
|
||||
"runwayml/gen4_aleph": {
|
||||
"runwayml/gen4.5": {
|
||||
"litellm_provider": "runwayml",
|
||||
"mode": "video_generation",
|
||||
"output_cost_per_video_per_second": 0.15,
|
||||
"output_cost_per_second": 0.12,
|
||||
"source": "https://docs.dev.runwayml.com/guides/pricing/",
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
|
|
@ -44055,13 +44081,136 @@
|
|||
"video"
|
||||
],
|
||||
"metadata": {
|
||||
"comment": "15 credits per second @ $0.01 per credit = $0.15 per second"
|
||||
"comment": "12 credits per second @ $0.01 per credit = $0.12 per second"
|
||||
}
|
||||
},
|
||||
"runwayml/gen3a_turbo": {
|
||||
"runwayml/aleph2": {
|
||||
"litellm_provider": "runwayml",
|
||||
"mode": "video_generation",
|
||||
"output_cost_per_video_per_second": 0.05,
|
||||
"output_cost_per_second": 0.28,
|
||||
"source": "https://docs.dev.runwayml.com/guides/pricing/",
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"video"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"video"
|
||||
],
|
||||
"metadata": {
|
||||
"comment": "28 credits per second @ $0.01 per credit = $0.28 per second; 56 credit minimum per task not modeled"
|
||||
}
|
||||
},
|
||||
"runwayml/seedance2": {
|
||||
"litellm_provider": "runwayml",
|
||||
"mode": "video_generation",
|
||||
"output_cost_per_second": 0.36,
|
||||
"output_cost_per_second_1080p": 0.4,
|
||||
"output_cost_per_second_4k": 1.5,
|
||||
"source": "https://docs.dev.runwayml.com/guides/pricing/",
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image",
|
||||
"video"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"video"
|
||||
],
|
||||
"metadata": {
|
||||
"comment": "36 credits per second at 480p/720p, 40 at 1080p, 150 at 4K @ $0.01 per credit"
|
||||
}
|
||||
},
|
||||
"runwayml/seedance2_fast": {
|
||||
"litellm_provider": "runwayml",
|
||||
"mode": "video_generation",
|
||||
"output_cost_per_second": 0.29,
|
||||
"source": "https://docs.dev.runwayml.com/guides/pricing/",
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image",
|
||||
"video"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"video"
|
||||
],
|
||||
"metadata": {
|
||||
"comment": "29 credits per second at 480p/720p @ $0.01 per credit = $0.29 per second"
|
||||
}
|
||||
},
|
||||
"runwayml/seedance2_mini": {
|
||||
"litellm_provider": "runwayml",
|
||||
"mode": "video_generation",
|
||||
"output_cost_per_second": 0.16,
|
||||
"source": "https://docs.dev.runwayml.com/guides/pricing/",
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image",
|
||||
"video"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"video"
|
||||
],
|
||||
"metadata": {
|
||||
"comment": "16 credits per second @ $0.01 per credit = $0.16 per second; 64 credit minimum per task not modeled"
|
||||
}
|
||||
},
|
||||
"runwayml/seedance2_5": {
|
||||
"litellm_provider": "runwayml",
|
||||
"mode": "video_generation",
|
||||
"output_cost_per_second": 0.3,
|
||||
"output_cost_per_second_480p": 0.2,
|
||||
"output_cost_per_second_1080p": 0.68,
|
||||
"source": "https://docs.dev.runwayml.com/guides/pricing/",
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image",
|
||||
"video"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"video"
|
||||
],
|
||||
"metadata": {
|
||||
"comment": "Output: 20/30/68 credits per second at 480p/720p/1080p @ $0.01 per credit; input video billed additionally at 10/15/34 credits per input second and the 80 credit minimum per task are not modeled"
|
||||
}
|
||||
},
|
||||
"runwayml/hailuo3": {
|
||||
"litellm_provider": "runwayml",
|
||||
"mode": "video_generation",
|
||||
"output_cost_per_second": 0.1,
|
||||
"output_cost_per_second_1080p": 0.15,
|
||||
"source": "https://docs.dev.runwayml.com/guides/pricing/",
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image",
|
||||
"video"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"video"
|
||||
],
|
||||
"metadata": {
|
||||
"comment": "10 credits per second at 768P, 15 at 2K (mapped to the 1080p tier) @ $0.01 per credit; 2 credits per reference image not modeled"
|
||||
}
|
||||
},
|
||||
"runwayml/gemini_omni_flash": {
|
||||
"litellm_provider": "runwayml",
|
||||
"mode": "video_generation",
|
||||
"output_cost_per_second": 0.1,
|
||||
"source": "https://docs.dev.runwayml.com/guides/pricing/",
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image",
|
||||
"video"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"video"
|
||||
],
|
||||
"metadata": {
|
||||
"comment": "10 credits per second @ $0.01 per credit = $0.10 per second"
|
||||
}
|
||||
},
|
||||
"runwayml/veo3.1": {
|
||||
"litellm_provider": "runwayml",
|
||||
"mode": "video_generation",
|
||||
"output_cost_per_second": 0.4,
|
||||
"source": "https://docs.dev.runwayml.com/guides/pricing/",
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
|
|
@ -44071,7 +44220,23 @@
|
|||
"video"
|
||||
],
|
||||
"metadata": {
|
||||
"comment": "5 credits per second @ $0.01 per credit = $0.05 per second"
|
||||
"comment": "40 credits per second with audio, 20 without @ $0.01 per credit; priced at the with-audio rate"
|
||||
}
|
||||
},
|
||||
"runwayml/veo3.1_fast": {
|
||||
"litellm_provider": "runwayml",
|
||||
"mode": "video_generation",
|
||||
"output_cost_per_second": 0.15,
|
||||
"source": "https://docs.dev.runwayml.com/guides/pricing/",
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"video"
|
||||
],
|
||||
"metadata": {
|
||||
"comment": "15 credits per second with audio, 10 without @ $0.01 per credit; priced at the with-audio rate"
|
||||
}
|
||||
},
|
||||
"runwayml/gen4_image": {
|
||||
|
|
@ -48728,6 +48893,7 @@
|
|||
"vertex_ai/claude-sonnet-4-6@default": {
|
||||
"regional_endpoint_uplift_multiplier": 1.1,
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"cache_creation_input_token_cost": 3.75e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 6e-06,
|
||||
"cache_read_input_token_cost": 3e-07,
|
||||
|
|
@ -49512,6 +49678,7 @@
|
|||
},
|
||||
"snowflake/claude-sonnet-4-6": {
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_legacy_thinking": true,
|
||||
"max_tokens": 16384,
|
||||
"max_input_tokens": 200000,
|
||||
"max_output_tokens": 16384,
|
||||
|
|
@ -50502,6 +50669,14 @@
|
|||
"supports_adaptive_thinking": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "claude-legacy-thinking",
|
||||
"pattern": "claude-[a-z]+-4[-._]6(?!\\d)",
|
||||
"description": "Claude at version 4.6 exactly, in any id shape that contains claude-<family>-4-6 (dotted and underscored minors included, dated releases such as claude-sonnet-4-6-20260219 too). The 4.6 family is adaptive-thinking yet still accepts legacy thinking.type=enabled with budget_tokens, so the caller's hard budget cap is forwarded verbatim instead of being rewritten to an uncapped output_config.effort. The lookahead keeps two-digit minors such as 4-60 from matching. 4.7+ and 5+ majors reject the legacy shape and stay on the adaptive translation.",
|
||||
"model_info": {
|
||||
"supports_legacy_thinking": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "claude-always-on-thinking",
|
||||
"pattern": "claude-(?:fable|mythos)-",
|
||||
|
|
|
|||
|
|
@ -428,6 +428,14 @@
|
|||
"type": "number",
|
||||
"minimum": 0
|
||||
},
|
||||
"output_cost_per_second_480p": {
|
||||
"type": "number",
|
||||
"minimum": 0
|
||||
},
|
||||
"output_cost_per_second_4k": {
|
||||
"type": "number",
|
||||
"minimum": 0
|
||||
},
|
||||
"output_cost_per_token": {
|
||||
"type": "number",
|
||||
"minimum": 0,
|
||||
|
|
@ -625,6 +633,9 @@
|
|||
"supports_image_size": {
|
||||
"type": "boolean"
|
||||
},
|
||||
"supports_legacy_thinking": {
|
||||
"type": "boolean"
|
||||
},
|
||||
"supports_low_reasoning_effort": {
|
||||
"type": "boolean"
|
||||
},
|
||||
|
|
|
|||
|
|
@ -57,7 +57,7 @@
|
|||
"limit": 3
|
||||
},
|
||||
"BLE001": {
|
||||
"limit": 2920
|
||||
"limit": 2919
|
||||
},
|
||||
"C401": {
|
||||
"limit": 8
|
||||
|
|
@ -108,7 +108,7 @@
|
|||
"limit": 3
|
||||
},
|
||||
"F401": {
|
||||
"limit": 17
|
||||
"limit": 14
|
||||
},
|
||||
"LOG015": {
|
||||
"limit": 5
|
||||
|
|
@ -152,9 +152,6 @@
|
|||
"PLW0127": {
|
||||
"limit": 57
|
||||
},
|
||||
"PLW0133": {
|
||||
"limit": 1
|
||||
},
|
||||
"PLW0602": {
|
||||
"limit": 215
|
||||
},
|
||||
|
|
|
|||
|
|
@ -40,6 +40,17 @@
|
|||
# later binding makes the name local for the whole body, so the read raises
|
||||
# UnboundLocalError, and in an autouse fixture that takes every test in the
|
||||
# directory down with it
|
||||
# F601 the same key literal twice in one dict. Python keeps the last value, so the
|
||||
# first is dropped before the test ever runs, and a fixture that looks like it
|
||||
# covers two cases covers one
|
||||
# B023 a closure over a loop variable. Every closure sees the last iteration's value,
|
||||
# so a per-case callback built in a loop checks the last case N times. Bind the
|
||||
# value as a parameter instead
|
||||
# B025 an `except` for a type an earlier `except` already catches. The second handler
|
||||
# is unreachable, so the recovery or skip written there never happens
|
||||
# F632 `is` against a literal. It compares identity, so it passes only where CPython
|
||||
# happens to intern the value and stops meaning what it says the moment the
|
||||
# value is built at runtime
|
||||
#
|
||||
# No target-version here on purpose: it resolves from requires-python (>=3.10), so
|
||||
# 3.11-only builtins like BaseExceptionGroup are correctly flagged in a tree that
|
||||
|
|
@ -63,4 +74,8 @@ lint.select = [
|
|||
"PLW0127",
|
||||
"RUF043",
|
||||
"F823",
|
||||
"F601",
|
||||
"B023",
|
||||
"B025",
|
||||
"F632",
|
||||
]
|
||||
|
|
|
|||
|
|
@ -5,9 +5,9 @@ lint.ignore = ["F405", "E402", "F403"]
|
|||
lint.extend-select = [
|
||||
"T20", "PGH004", "RUF008", "RUF009", "RUF100",
|
||||
"B033", "FURB136", "FURB168", "FURB188", "I001", "PERF402", "PIE790", "PIE800", "PLC0208",
|
||||
"PLR0402", "PLR1711", "PLR1730", "PLR2044", "PYI030", "PYI041", "PYI064", "RET501", "RUF010",
|
||||
"RUF022", "RUF023", "RUF051", "SIM114", "SIM118", "TC005", "UP006", "UP007", "UP008", "UP012",
|
||||
"UP018", "UP024", "UP032", "UP034", "UP035", "UP037", "UP045",
|
||||
"PLR0402", "PLR1711", "PLR1730", "PLR2044", "PLW0133", "PYI030", "PYI041", "PYI064", "RET501",
|
||||
"RUF010", "RUF022", "RUF023", "RUF051", "SIM114", "SIM118", "TC005", "UP006", "UP007", "UP008",
|
||||
"UP012", "UP018", "UP024", "UP032", "UP034", "UP035", "UP037", "UP045",
|
||||
]
|
||||
# RUF100 (unused-noqa) only knows the rules enabled in THIS config, so it would strip
|
||||
# `# noqa` directives that protect rules enforced elsewhere. List those codes as external
|
||||
|
|
|
|||
|
|
@ -93,7 +93,7 @@ def get_bedrock_pricing(url, providers):
|
|||
else:
|
||||
# General logic for other providers
|
||||
section = soup.find(
|
||||
"h2", text=lambda t: t and provider.lower() in t.lower()
|
||||
"h2", text=lambda t, needle=provider.lower(): t and needle in t.lower()
|
||||
)
|
||||
if not section:
|
||||
pricing_data[provider] = "Provider section not found"
|
||||
|
|
|
|||
|
|
@ -64,11 +64,6 @@ def test_langsmith_logging_async():
|
|||
except Exception as e:
|
||||
pytest.fail(f"An exception occurred - {e}")
|
||||
|
||||
except litellm.Timeout as e:
|
||||
pass
|
||||
except Exception as e:
|
||||
pytest.fail(f"An exception occurred - {e}")
|
||||
|
||||
|
||||
async def make_async_calls(metadata=None, **completion_kwargs):
|
||||
total_tasks = 300
|
||||
|
|
|
|||
|
|
@ -4202,13 +4202,7 @@ def test_gemini_google_maps_tool_simple():
|
|||
)
|
||||
print(f"Response: {response.model_dump_json(indent=4)}")
|
||||
assert response.choices[0].message.content is not None
|
||||
except (litellm.RateLimitError, litellm.InternalServerError):
|
||||
# Transient Vertex-side failures (rate limiting, 500 INTERNAL from the
|
||||
# Google Maps grounding backend) are not LiteLLM bugs — don't fail CI.
|
||||
pass
|
||||
except litellm.InternalServerError:
|
||||
pytest.skip(
|
||||
"Google Maps Platform returned a transient 500 (upstream flake); skipping."
|
||||
)
|
||||
except (litellm.RateLimitError, litellm.InternalServerError) as e:
|
||||
pytest.skip(f"Transient Vertex-side failure, not a LiteLLM bug: {e}")
|
||||
except Exception as e:
|
||||
pytest.fail(f"Error occurred: {e}")
|
||||
|
|
|
|||
|
|
@ -143,7 +143,6 @@ def test_spend_logs_payload(model_id: Optional[str]):
|
|||
"completion_start_time": datetime.datetime(2024, 6, 7, 12, 43, 30, 954146),
|
||||
"max_tokens": 10,
|
||||
"extra_body": {},
|
||||
"custom_llm_provider": "azure",
|
||||
"input": [
|
||||
{"role": "system", "content": "you are a helpful assistant.\n"},
|
||||
{"role": "user", "content": "bom dia"},
|
||||
|
|
|
|||
|
|
@ -3323,16 +3323,17 @@ async def test_team_access_groups(prisma_client):
|
|||
|
||||
request._url = URL(url="/chat/completions")
|
||||
|
||||
def body_reader(requested_model: str):
|
||||
async def return_body() -> bytes:
|
||||
return f'{{"model": "{requested_model}"}}'.encode()
|
||||
|
||||
return return_body
|
||||
|
||||
for model in ["gpt-4o", "gemini-pro-vision"]:
|
||||
# Expect these to pass
|
||||
async def return_body():
|
||||
return_string = f'{{"model": "{model}"}}'
|
||||
# return string as bytes
|
||||
return return_string.encode()
|
||||
|
||||
request = Request(scope={"type": "http"})
|
||||
request._url = URL(url="/chat/completions")
|
||||
request.body = return_body
|
||||
request.body = body_reader(model)
|
||||
|
||||
# use generated key to auth in
|
||||
print(
|
||||
|
|
@ -3342,14 +3343,9 @@ async def test_team_access_groups(prisma_client):
|
|||
|
||||
for model in ["gpt-4", "gpt-4o-mini", "gemini-experimental"]:
|
||||
# Expect these to fail
|
||||
async def return_body_2():
|
||||
return_string = f'{{"model": "{model}"}}'
|
||||
# return string as bytes
|
||||
return return_string.encode()
|
||||
|
||||
request = Request(scope={"type": "http"})
|
||||
request._url = URL(url="/chat/completions")
|
||||
request.body = return_body_2
|
||||
request.body = body_reader(model)
|
||||
|
||||
# use generated key to auth in
|
||||
print(
|
||||
|
|
|
|||
|
|
@ -867,6 +867,7 @@ PROVIDERS_WITH_A_HANDLER = (
|
|||
"openrouter",
|
||||
"perplexity",
|
||||
"replicate",
|
||||
"runwayml",
|
||||
"sagemaker",
|
||||
"together_ai",
|
||||
"vertex_ai",
|
||||
|
|
@ -956,6 +957,7 @@ PROVIDERS_THAT_RECOGNISE_A_FULL_CONTEXT_WINDOW = (
|
|||
"mistral",
|
||||
"openai",
|
||||
"perplexity",
|
||||
"runwayml",
|
||||
"together_ai",
|
||||
"vertex_ai",
|
||||
"xai",
|
||||
|
|
@ -971,6 +973,7 @@ PROVIDERS_THAT_RECOGNISE_A_CONTENT_POLICY_BLOCK = (
|
|||
"mistral",
|
||||
"openai",
|
||||
"perplexity",
|
||||
"runwayml",
|
||||
"together_ai",
|
||||
"xai",
|
||||
)
|
||||
|
|
|
|||
|
|
@ -2,7 +2,6 @@
|
|||
|
||||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.constants import (
|
||||
DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET,
|
||||
DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET,
|
||||
|
|
@ -17,7 +16,6 @@ from litellm.llms.bedrock.messages.invoke_transformations.anthropic_claude3_tran
|
|||
)
|
||||
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"reasoning_effort,expected_effort",
|
||||
[
|
||||
|
|
@ -258,19 +256,22 @@ def test_reasoning_effort_in_supported_params():
|
|||
"model",
|
||||
[
|
||||
"claude-sonnet-4-6",
|
||||
"bedrock/invoke/us.anthropic.claude-sonnet-4-6",
|
||||
"vertex_ai/claude-sonnet-4-6",
|
||||
"claude-opus-4-6",
|
||||
"claude-sonnet-4-6-20260219",
|
||||
"bedrock/invoke/us.anthropic.claude-sonnet-4-6",
|
||||
"bedrock/invoke/us.anthropic.claude-opus-4-6-v1:0",
|
||||
"vertex_ai/claude-sonnet-4-6",
|
||||
"vertex_ai/claude-opus-4-6",
|
||||
"azure_ai/claude-sonnet-4-6",
|
||||
],
|
||||
)
|
||||
def test_legacy_thinking_high_budget_clamps_to_high_when_xhigh_unsupported(
|
||||
local_model_cost_map, model
|
||||
):
|
||||
"""Claude Code sends ``thinking.budget_tokens=31999``; Sonnet 4.6 and Opus 4.6
|
||||
have no ``xhigh`` tier, so the translator must emit ``high`` rather than the
|
||||
provider-invalid ``xhigh`` (regression for issue #29282)."""
|
||||
def test_legacy_thinking_budget_preserved_verbatim_on_46(local_model_cost_map, model):
|
||||
"""Regression for the passthrough silently dropping a caller's hard thinking
|
||||
budget: the 4.6 family accepts ``thinking.type=enabled`` with ``budget_tokens``
|
||||
natively, so rewriting it to ``thinking.type=adaptive`` + ``output_config.effort``
|
||||
(which carries no ceiling) let reasoning run past the requested cap. The legacy
|
||||
shape must be forwarded verbatim, in every 4.6 id shape including unmapped dated
|
||||
releases resolved by the ``claude-legacy-thinking`` fallback rule."""
|
||||
config = AnthropicMessagesConfig()
|
||||
optional_params = {
|
||||
"max_tokens": 1024,
|
||||
|
|
@ -285,8 +286,8 @@ def test_legacy_thinking_high_budget_clamps_to_high_when_xhigh_unsupported(
|
|||
headers={},
|
||||
)
|
||||
|
||||
assert result.get("thinking") == {"type": "adaptive"}
|
||||
assert result.get("output_config") == {"effort": "high"}
|
||||
assert result.get("thinking") == {"type": "enabled", "budget_tokens": 31999}
|
||||
assert "output_config" not in result
|
||||
|
||||
|
||||
def test_legacy_thinking_high_budget_keeps_xhigh_when_supported():
|
||||
|
|
@ -343,11 +344,44 @@ def test_legacy_thinking_translates_to_adaptive_for_opus_48(
|
|||
assert result.get("output_config") == {"effort": "xhigh"}
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model,expected_effort",
|
||||
[
|
||||
("claude-sonnet-5", "xhigh"),
|
||||
("claude-opus-5", "xhigh"),
|
||||
("claude-newfamily-6", "high"),
|
||||
],
|
||||
)
|
||||
def test_legacy_thinking_translates_to_adaptive_for_5_and_future_models(
|
||||
local_model_cost_map, model, expected_effort
|
||||
):
|
||||
"""The 5 families reject ``thinking.type=enabled``, so the adaptive translation
|
||||
stays the safe default for every adaptive model not flagged
|
||||
``supports_legacy_thinking``, unmapped future ids included. An unmapped id
|
||||
cannot prove ``xhigh`` support, so its high-budget bucket clamps to ``high``."""
|
||||
config = AnthropicMessagesConfig()
|
||||
optional_params = {
|
||||
"max_tokens": 1024,
|
||||
"thinking": {"type": "enabled", "budget_tokens": 31999},
|
||||
}
|
||||
|
||||
result = config.transform_anthropic_messages_request(
|
||||
model=model,
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
anthropic_messages_optional_request_params=optional_params,
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert result.get("thinking") == {"type": "adaptive"}
|
||||
assert result.get("output_config") == {"effort": expected_effort}
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"budget_tokens,expected_effort",
|
||||
[
|
||||
(DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET * 2, "high"),
|
||||
(DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET, "high"),
|
||||
(DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET * 2, "xhigh"),
|
||||
(DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET, "xhigh"),
|
||||
(DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET, "high"),
|
||||
(DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET - 1, "medium"),
|
||||
(DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET, "medium"),
|
||||
|
|
@ -355,7 +389,9 @@ def test_legacy_thinking_translates_to_adaptive_for_opus_48(
|
|||
(1, "low"),
|
||||
],
|
||||
)
|
||||
def test_legacy_thinking_budget_buckets_on_sonnet_46(budget_tokens, expected_effort):
|
||||
def test_legacy_thinking_budget_buckets_on_opus_48(
|
||||
local_model_cost_map, budget_tokens, expected_effort
|
||||
):
|
||||
config = AnthropicMessagesConfig()
|
||||
optional_params = {
|
||||
"max_tokens": 1024,
|
||||
|
|
@ -363,7 +399,7 @@ def test_legacy_thinking_budget_buckets_on_sonnet_46(budget_tokens, expected_eff
|
|||
}
|
||||
|
||||
result = config.transform_anthropic_messages_request(
|
||||
model="claude-sonnet-4-6",
|
||||
model="claude-opus-4-8",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
anthropic_messages_optional_request_params=optional_params,
|
||||
litellm_params={},
|
||||
|
|
@ -373,7 +409,29 @@ def test_legacy_thinking_budget_buckets_on_sonnet_46(budget_tokens, expected_eff
|
|||
assert result.get("output_config") == {"effort": expected_effort}
|
||||
|
||||
|
||||
def test_legacy_thinking_does_not_override_explicit_output_config():
|
||||
def test_legacy_thinking_does_not_override_explicit_output_config(local_model_cost_map):
|
||||
config = AnthropicMessagesConfig()
|
||||
optional_params = {
|
||||
"max_tokens": 1024,
|
||||
"thinking": {"type": "enabled", "budget_tokens": 31999},
|
||||
"output_config": {"effort": "low"},
|
||||
}
|
||||
|
||||
result = config.transform_anthropic_messages_request(
|
||||
model="claude-opus-4-8",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
anthropic_messages_optional_request_params=optional_params,
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert result.get("thinking") == {"type": "adaptive"}
|
||||
assert result.get("output_config") == {"effort": "low"}
|
||||
|
||||
|
||||
def test_legacy_thinking_with_explicit_output_config_untouched_on_46(
|
||||
local_model_cost_map,
|
||||
):
|
||||
config = AnthropicMessagesConfig()
|
||||
optional_params = {
|
||||
"max_tokens": 1024,
|
||||
|
|
@ -389,6 +447,7 @@ def test_legacy_thinking_does_not_override_explicit_output_config():
|
|||
headers={},
|
||||
)
|
||||
|
||||
assert result.get("thinking") == {"type": "enabled", "budget_tokens": 31999}
|
||||
assert result.get("output_config") == {"effort": "low"}
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -475,7 +475,7 @@ class TestOllamaTextCompletionResponseIterator:
|
|||
assert isinstance(result, ModelResponseStream)
|
||||
assert result.choices and result.choices[0].delta is not None
|
||||
assert result.choices[0].delta.content == None
|
||||
assert getattr(result.choices[0].delta, "reasoning_content", None) is ""
|
||||
assert getattr(result.choices[0].delta, "reasoning_content", None) == ""
|
||||
|
||||
def test_chunk_parser_done_chunk(self):
|
||||
"""Test that done chunks work correctly."""
|
||||
|
|
|
|||
|
|
@ -7,7 +7,12 @@ from unittest.mock import Mock
|
|||
import httpx
|
||||
import pytest
|
||||
|
||||
from litellm.llms.runwayml.videos.transformation import RunwayMLVideoConfig
|
||||
from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
||||
from litellm.llms.runwayml.videos.transformation import (
|
||||
RunwayMLError,
|
||||
RunwayMLVideoConfig,
|
||||
_ratio_to_resolution,
|
||||
)
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
from litellm.types.videos.main import VideoObject
|
||||
|
||||
|
|
@ -49,6 +54,158 @@ class TestRunwayMLVideoTransformation:
|
|||
# Validate URL has correct endpoint
|
||||
assert url == "https://api.dev.runwayml.com/v1/image_to_video"
|
||||
|
||||
def test_transform_video_create_request_text_to_video(self):
|
||||
"""A prompt-only request must hit /text_to_video, not /image_to_video."""
|
||||
data, files, url = self.config.transform_video_create_request(
|
||||
model="veo3.1",
|
||||
prompt="A serene mountain lake at sunrise",
|
||||
api_base="https://api.dev.runwayml.com/v1",
|
||||
video_create_optional_request_params={"duration": 8, "ratio": "1280:720"},
|
||||
litellm_params=GenericLiteLLMParams(),
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert url == "https://api.dev.runwayml.com/v1/text_to_video"
|
||||
assert "promptImage" not in data
|
||||
assert data["promptText"] == "A serene mountain lake at sunrise"
|
||||
|
||||
def test_transform_video_create_request_video_to_video(self):
|
||||
"""A promptVideo request must hit /video_to_video with promptImage stripped."""
|
||||
data, files, url = self.config.transform_video_create_request(
|
||||
model="aleph2",
|
||||
prompt="Make it snow",
|
||||
api_base="https://api.dev.runwayml.com/v1",
|
||||
video_create_optional_request_params={
|
||||
"promptVideo": "https://example.com/source.mp4",
|
||||
"promptImage": "https://example.com/reference.png",
|
||||
"ratio": "1280:720",
|
||||
},
|
||||
litellm_params=GenericLiteLLMParams(),
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert url == "https://api.dev.runwayml.com/v1/video_to_video"
|
||||
assert data["promptVideo"] == "https://example.com/source.mp4"
|
||||
assert "promptImage" not in data
|
||||
|
||||
def test_transform_video_create_request_video_uri_routes_to_video_to_video(self):
|
||||
_, _, url = self.config.transform_video_create_request(
|
||||
model="aleph2",
|
||||
prompt="Make it snow",
|
||||
api_base="https://api.dev.runwayml.com/v1",
|
||||
video_create_optional_request_params={"videoUri": "https://example.com/source.mp4"},
|
||||
litellm_params=GenericLiteLLMParams(),
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert url == "https://api.dev.runwayml.com/v1/video_to_video"
|
||||
|
||||
def test_status_progress_fraction_scales_to_percent(self):
|
||||
"""Runway reports progress as a 0..1 float; VideoObject.progress is an int percent."""
|
||||
mock_response = Mock(spec=httpx.Response)
|
||||
mock_response.json.return_value = {
|
||||
"id": "63fd0f13-f29d-4e58-99d3-1cb9efa14a5b",
|
||||
"createdAt": "2025-11-11T21:48:50.448Z",
|
||||
"status": "RUNNING",
|
||||
"progress": 0.027,
|
||||
}
|
||||
|
||||
result = self.config.transform_video_status_retrieve_response(
|
||||
raw_response=mock_response,
|
||||
logging_obj=self.mock_logging_obj,
|
||||
custom_llm_provider="runwayml",
|
||||
)
|
||||
|
||||
assert result.status == "in_progress"
|
||||
assert result.progress == 3
|
||||
|
||||
def test_status_progress_null_leaves_progress_unset(self):
|
||||
"""Runway sends an explicit null progress for pending polls; scaling it must not crash."""
|
||||
mock_response = Mock(spec=httpx.Response)
|
||||
mock_response.json.return_value = {
|
||||
"id": "63fd0f13-f29d-4e58-99d3-1cb9efa14a5b",
|
||||
"createdAt": "2025-11-11T21:48:50.448Z",
|
||||
"status": "PENDING",
|
||||
"progress": None,
|
||||
}
|
||||
|
||||
result = self.config.transform_video_status_retrieve_response(
|
||||
raw_response=mock_response,
|
||||
logging_obj=self.mock_logging_obj,
|
||||
custom_llm_provider="runwayml",
|
||||
)
|
||||
|
||||
assert result.status == "queued"
|
||||
assert result.progress is None
|
||||
|
||||
def test_get_error_class_returns_exception_instead_of_raising(self):
|
||||
error = self.config.get_error_class(
|
||||
error_message="Invalid API key",
|
||||
status_code=401,
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert isinstance(error, RunwayMLError)
|
||||
assert isinstance(error, BaseLLMException)
|
||||
assert error.status_code == 401
|
||||
assert error.message == "Invalid API key"
|
||||
|
||||
def test_create_response_usage_includes_resolution_and_provider_cost(self):
|
||||
mock_response = Mock(spec=httpx.Response)
|
||||
mock_response.json.return_value = {
|
||||
"id": "test-video-id-123",
|
||||
"createdAt": "2025-11-11T21:48:50.448Z",
|
||||
"status": "PENDING",
|
||||
"estimatedCost": {"credits": 25.0},
|
||||
}
|
||||
|
||||
video_obj = self.config.transform_video_create_response(
|
||||
model="gen4_turbo",
|
||||
raw_response=mock_response,
|
||||
logging_obj=self.mock_logging_obj,
|
||||
custom_llm_provider="runwayml",
|
||||
request_data={"model": "gen4_turbo", "ratio": "1280:720", "duration": 5},
|
||||
)
|
||||
|
||||
assert video_obj.usage == {
|
||||
"duration_seconds": 5.0,
|
||||
"video_resolution": "720p",
|
||||
"provider_reported_cost_usd": 0.25,
|
||||
}
|
||||
|
||||
def test_create_response_usage_omits_unknown_fields(self):
|
||||
mock_response = Mock(spec=httpx.Response)
|
||||
mock_response.json.return_value = {
|
||||
"id": "test-video-id-123",
|
||||
"createdAt": "2025-11-11T21:48:50.448Z",
|
||||
"status": "PENDING",
|
||||
}
|
||||
|
||||
video_obj = self.config.transform_video_create_response(
|
||||
model="gen4_turbo",
|
||||
raw_response=mock_response,
|
||||
logging_obj=self.mock_logging_obj,
|
||||
custom_llm_provider="runwayml",
|
||||
request_data={"model": "gen4_turbo"},
|
||||
)
|
||||
|
||||
assert video_obj.usage == {}
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"ratio,expected",
|
||||
[
|
||||
("848:480", "480p"),
|
||||
("1280:720", "720p"),
|
||||
("1920:1080", "1080p"),
|
||||
("2560:1440", "1080p"),
|
||||
("3840:2160", "4k"),
|
||||
(None, None),
|
||||
("banana", None),
|
||||
],
|
||||
)
|
||||
def test_ratio_to_resolution_tiers(self, ratio, expected):
|
||||
assert _ratio_to_resolution(ratio) == expected
|
||||
|
||||
def test_transform_video_status_with_timestamp_handling(self):
|
||||
"""Test status retrieval handles RunwayML's ISO 8601 timestamps correctly."""
|
||||
from litellm.types.videos.utils import encode_video_id_with_provider
|
||||
|
|
|
|||
|
|
@ -124,11 +124,7 @@ class TestGenerateIAMToken:
|
|||
mock_client.reset_mock()
|
||||
mock_cache.reset_mock()
|
||||
|
||||
# Configure mock to return values based on env_keys
|
||||
def get_secret_side_effect(key):
|
||||
return env_keys.get(key)
|
||||
|
||||
mock_get_secret_str.side_effect = get_secret_side_effect
|
||||
mock_get_secret_str.side_effect = env_keys.get
|
||||
|
||||
mock_response = MagicMock()
|
||||
mock_response.json.return_value = {
|
||||
|
|
|
|||
|
|
@ -18,6 +18,7 @@ from unittest.mock import AsyncMock, MagicMock
|
|||
|
||||
import orjson
|
||||
import pytest
|
||||
from starlette.datastructures import FormData
|
||||
|
||||
from litellm.ocr.main import convert_file_document_to_url_document, get_mime_type
|
||||
|
||||
|
|
@ -470,7 +471,7 @@ class TestProxySecurityGuard:
|
|||
|
||||
mock_request = MagicMock()
|
||||
mock_request.headers = {"content-type": "multipart/form-data; boundary=---"}
|
||||
mock_request.form = AsyncMock(return_value=mock_form)
|
||||
mock_request.form = AsyncMock(return_value=FormData(mock_form))
|
||||
|
||||
result = await self._parse_multipart(mock_request)
|
||||
|
||||
|
|
|
|||
|
|
@ -5,6 +5,7 @@ import orjson
|
|||
import pytest
|
||||
from fastapi import Request
|
||||
from fastapi.testclient import TestClient
|
||||
from starlette.datastructures import FormData
|
||||
|
||||
|
||||
|
||||
|
|
@ -68,7 +69,7 @@ async def test_form_data_parsing():
|
|||
test_data = {"name": "test_user", "message": "hello world"}
|
||||
|
||||
# Mock the form method to return the test data as an awaitable
|
||||
mock_request.form = AsyncMock(return_value=test_data)
|
||||
mock_request.form = AsyncMock(return_value=FormData(test_data))
|
||||
mock_request.headers = {"content-type": "application/x-www-form-urlencoded"}
|
||||
mock_request.scope = {}
|
||||
mock_request.state._cached_headers = None
|
||||
|
|
@ -119,7 +120,7 @@ async def test_form_data_with_json_metadata():
|
|||
}
|
||||
|
||||
# Mock the form method to return the test data as an awaitable
|
||||
mock_request.form = AsyncMock(return_value=test_data)
|
||||
mock_request.form = AsyncMock(return_value=FormData(test_data))
|
||||
mock_request.headers = {"content-type": "multipart/form-data"}
|
||||
mock_request.scope = {}
|
||||
mock_request.state._cached_headers = None
|
||||
|
|
@ -160,7 +161,7 @@ async def test_form_data_with_invalid_json_metadata():
|
|||
}
|
||||
|
||||
# Mock the form method to return the test data
|
||||
mock_request.form = AsyncMock(return_value=test_data)
|
||||
mock_request.form = AsyncMock(return_value=FormData(test_data))
|
||||
mock_request.headers = {"content-type": "multipart/form-data"}
|
||||
mock_request.scope = {}
|
||||
mock_request.state._cached_headers = None
|
||||
|
|
@ -183,7 +184,7 @@ async def test_form_data_without_metadata():
|
|||
test_data = {"model": "whisper-1", "file": "audio.mp3", "language": "en"}
|
||||
|
||||
# Mock the form method to return the test data
|
||||
mock_request.form = AsyncMock(return_value=test_data)
|
||||
mock_request.form = AsyncMock(return_value=FormData(test_data))
|
||||
mock_request.headers = {"content-type": "application/x-www-form-urlencoded"}
|
||||
mock_request.scope = {}
|
||||
mock_request.state._cached_headers = None
|
||||
|
|
@ -214,7 +215,7 @@ async def test_form_data_with_empty_metadata():
|
|||
}
|
||||
|
||||
# Mock the form method to return the test data
|
||||
mock_request.form = AsyncMock(return_value=test_data)
|
||||
mock_request.form = AsyncMock(return_value=FormData(test_data))
|
||||
mock_request.headers = {"content-type": "multipart/form-data"}
|
||||
mock_request.scope = {}
|
||||
mock_request.state._cached_headers = None
|
||||
|
|
@ -249,7 +250,7 @@ async def test_form_data_with_dict_metadata():
|
|||
}
|
||||
|
||||
# Mock the form method to return the test data
|
||||
mock_request.form = AsyncMock(return_value=test_data)
|
||||
mock_request.form = AsyncMock(return_value=FormData(test_data))
|
||||
mock_request.headers = {"content-type": "multipart/form-data"}
|
||||
mock_request.scope = {}
|
||||
mock_request.state._cached_headers = None
|
||||
|
|
@ -280,7 +281,7 @@ async def test_form_data_with_none_metadata():
|
|||
}
|
||||
|
||||
# Mock the form method to return the test data
|
||||
mock_request.form = AsyncMock(return_value=test_data)
|
||||
mock_request.form = AsyncMock(return_value=FormData(test_data))
|
||||
mock_request.headers = {"content-type": "multipart/form-data"}
|
||||
mock_request.scope = {}
|
||||
mock_request.state._cached_headers = None
|
||||
|
|
@ -495,33 +496,29 @@ async def test_surrogate_repair_skipped_above_size_limit(monkeypatch):
|
|||
@pytest.mark.asyncio
|
||||
async def test_get_form_data():
|
||||
"""
|
||||
Test that get_form_data correctly handles form data with array notation.
|
||||
Tests audio transcription parameters as a specific example.
|
||||
A repeated `foo[]` key is how the OpenAI SDKs send a list, so every value has to
|
||||
survive. `FormData`, not a dict: a dict cannot even hold the duplicate key.
|
||||
"""
|
||||
# Create a mock request with transcription form data
|
||||
mock_request = MagicMock()
|
||||
mock_request.form = AsyncMock(
|
||||
return_value=FormData(
|
||||
[
|
||||
("file", "file_object"),
|
||||
("model", "gpt-4o-transcribe"),
|
||||
("include[]", "logprobs"),
|
||||
("language", "en"),
|
||||
("prompt", "Transcribe this audio file"),
|
||||
("response_format", "json"),
|
||||
("stream", "false"),
|
||||
("temperature", "0.2"),
|
||||
("timestamp_granularities[]", "word"),
|
||||
("timestamp_granularities[]", "segment"),
|
||||
]
|
||||
)
|
||||
)
|
||||
|
||||
# Create mock form data with array notation for timestamp_granularities
|
||||
mock_form_data = {
|
||||
"file": "file_object", # In a real request this would be an UploadFile
|
||||
"model": "gpt-4o-transcribe",
|
||||
"include[]": "logprobs", # Array notation
|
||||
"language": "en",
|
||||
"prompt": "Transcribe this audio file",
|
||||
"response_format": "json",
|
||||
"stream": "false",
|
||||
"temperature": "0.2",
|
||||
"timestamp_granularities[]": "word", # First array item
|
||||
"timestamp_granularities[]": "segment", # Second array item (would overwrite in dict, but handled by the function)
|
||||
}
|
||||
|
||||
# Mock the form method to return the test data
|
||||
mock_request.form = AsyncMock(return_value=mock_form_data)
|
||||
|
||||
# Call the function being tested
|
||||
result = await get_form_data(mock_request)
|
||||
|
||||
# Verify regular form fields are preserved
|
||||
assert result["file"] == "file_object"
|
||||
assert result["model"] == "gpt-4o-transcribe"
|
||||
assert result["language"] == "en"
|
||||
|
|
@ -529,17 +526,8 @@ async def test_get_form_data():
|
|||
assert result["response_format"] == "json"
|
||||
assert result["stream"] == "false"
|
||||
assert result["temperature"] == "0.2"
|
||||
|
||||
# Verify array fields are correctly parsed
|
||||
assert "include" in result
|
||||
assert isinstance(result["include"], list)
|
||||
assert "logprobs" in result["include"]
|
||||
|
||||
assert "timestamp_granularities" in result
|
||||
assert isinstance(result["timestamp_granularities"], list)
|
||||
# Note: In a real MultiDict, both values would be present
|
||||
# But in our mock dictionary the second value overwrites the first
|
||||
assert "segment" in result["timestamp_granularities"]
|
||||
assert result["include"] == ["logprobs"]
|
||||
assert result["timestamp_granularities"] == ["word", "segment"]
|
||||
|
||||
|
||||
def test_get_tags_from_request_body_with_metadata_tags():
|
||||
|
|
@ -953,7 +941,7 @@ class TestReadRequestBodyNonCanonicalContentType:
|
|||
|
||||
mock_request = MagicMock()
|
||||
mock_request.body = AsyncMock(return_value=orjson.dumps(payload))
|
||||
mock_request.form = AsyncMock(return_value={})
|
||||
mock_request.form = AsyncMock(return_value=FormData({}))
|
||||
mock_request.headers = {"content-type": content_type}
|
||||
mock_request.scope = {}
|
||||
|
||||
|
|
@ -964,7 +952,7 @@ class TestReadRequestBodyNonCanonicalContentType:
|
|||
@pytest.mark.asyncio
|
||||
async def test_real_form_post_still_parsed_as_form(self):
|
||||
mock_request = MagicMock()
|
||||
mock_request.form = AsyncMock(return_value={"k": "v"})
|
||||
mock_request.form = AsyncMock(return_value=FormData({"k": "v"}))
|
||||
mock_request.body = AsyncMock(return_value=b"")
|
||||
mock_request.headers = {"content-type": "application/x-www-form-urlencoded"}
|
||||
mock_request.scope = {}
|
||||
|
|
@ -1020,7 +1008,7 @@ class TestGetRequestBody:
|
|||
mock_request = MagicMock()
|
||||
mock_request.method = "POST"
|
||||
mock_request.headers = {"content-type": "multipart/form-data; boundary=x"}
|
||||
mock_request.form = AsyncMock(return_value={"k": "v"})
|
||||
mock_request.form = AsyncMock(return_value=FormData({"k": "v"}))
|
||||
mock_request.scope = {}
|
||||
|
||||
result = await get_request_body(mock_request)
|
||||
|
|
|
|||
|
|
@ -898,6 +898,62 @@ async def test_test_model_connection_uses_loaded_deployment_team_id_via_model_na
|
|||
assert passed_model_params.model_info.team_id == deployment_owner_team_id
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_test_model_connection_authorizes_on_params_after_health_check_params_merge():
|
||||
"""
|
||||
Regression guard for the ordering fix: health_check_params from the request
|
||||
body are merged into the probe params BEFORE the authorization check, so a
|
||||
caller cannot smuggle a field past auth via health_check_params. Auth is
|
||||
stubbed to reject, which halts the endpoint right after it records the
|
||||
params it was handed, so the outbound probe is never reached. If the merge
|
||||
is moved back to after can_user_make_model_call, the marker is absent from
|
||||
those params and this test fails.
|
||||
"""
|
||||
from fastapi import HTTPException
|
||||
|
||||
from litellm.proxy.management_endpoints.model_management_endpoints import (
|
||||
ModelManagementAuthChecks,
|
||||
)
|
||||
from litellm.types.router import Deployment
|
||||
|
||||
marker = "sentinel-from-health-check-params"
|
||||
mock_can_user_make_model_call = AsyncMock(
|
||||
side_effect=HTTPException(status_code=403, detail="denied")
|
||||
)
|
||||
|
||||
with (
|
||||
patch( # test-quality-ok: proxy module global, no injection seam
|
||||
"litellm.proxy.proxy_server.prisma_client", MagicMock()
|
||||
),
|
||||
patch( # test-quality-ok: proxy module global, no injection seam
|
||||
"litellm.proxy.proxy_server.llm_router", None
|
||||
),
|
||||
patch.object( # test-quality-ok: capturing the params handed to auth is the assertion
|
||||
ModelManagementAuthChecks,
|
||||
"can_user_make_model_call",
|
||||
mock_can_user_make_model_call,
|
||||
),
|
||||
pytest.raises(HTTPException),
|
||||
):
|
||||
await health_test_model_connection(
|
||||
request=MagicMock(),
|
||||
mode="chat",
|
||||
litellm_params={"model": "openai/gpt-4o"},
|
||||
model_info={"health_check_params": {"probe_marker": marker}},
|
||||
user_api_key_dict=UserAPIKeyAuth(
|
||||
token="requester-token",
|
||||
user_id="admin-user",
|
||||
user_role=LitellmUserRoles.PROXY_ADMIN,
|
||||
),
|
||||
)
|
||||
|
||||
assert mock_can_user_make_model_call.called
|
||||
passed_model_params = mock_can_user_make_model_call.call_args.kwargs["model_params"]
|
||||
assert isinstance(passed_model_params, Deployment)
|
||||
authorized_params = passed_model_params.litellm_params.model_dump()
|
||||
assert authorized_params.get("probe_marker") == marker
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_test_model_connection_authorized_team_admin_passes_real_auth():
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -12,6 +12,7 @@ import httpx
|
|||
import pytest
|
||||
from fastapi import HTTPException, Request, Response
|
||||
from fastapi.testclient import TestClient
|
||||
from starlette.datastructures import FormData
|
||||
|
||||
|
||||
import litellm
|
||||
|
|
@ -1380,7 +1381,7 @@ async def test_is_streaming_request_fn():
|
|||
mock_request = Mock()
|
||||
mock_request.method = "POST"
|
||||
mock_request.headers = {"content-type": "multipart/form-data"}
|
||||
mock_request.form = AsyncMock(return_value={"stream": "true"})
|
||||
mock_request.form = AsyncMock(return_value=FormData({"stream": "true"}))
|
||||
assert await is_streaming_request_fn(mock_request) is True
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -1,7 +1,11 @@
|
|||
import json
|
||||
import logging
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
import pytest
|
||||
import respx
|
||||
|
||||
import litellm
|
||||
from litellm.litellm_core_utils.health_check_helpers import HealthCheckHelpers
|
||||
from litellm.proxy import health_check as hc_module
|
||||
from litellm.proxy.health_check import (
|
||||
|
|
@ -534,6 +538,135 @@ async def test_run_model_health_check_skips_auto_router_deployment():
|
|||
assert result == {}
|
||||
|
||||
|
||||
def test_health_check_params_merge_into_probe_params():
|
||||
"""health_check_params reach the probe request for the deployment that declares them."""
|
||||
media_source = {"s3Location": {"uri": "s3://my-bucket/clip.mp4"}}
|
||||
|
||||
updated = _update_litellm_params_for_health_check(
|
||||
{"mode": "chat", "health_check_params": {"mediaSource": media_source}},
|
||||
{"model": "bedrock/us.twelvelabs.pegasus-1-2-v1:0"},
|
||||
)
|
||||
|
||||
assert updated["mediaSource"] == media_source
|
||||
assert updated["model"] == "us.twelvelabs.pegasus-1-2-v1:0"
|
||||
assert updated["custom_llm_provider"] == "bedrock"
|
||||
|
||||
|
||||
def test_health_check_params_lose_to_dedicated_health_check_knobs():
|
||||
"""The dedicated knobs are applied after the merge, so they win on conflict."""
|
||||
model_info = {
|
||||
"mode": "chat",
|
||||
"health_check_params": {
|
||||
"max_tokens": 4096,
|
||||
"model": "openai/expensive-model",
|
||||
"messages": [{"role": "user", "content": "from health_check_params"}],
|
||||
"reasoning_effort": "high",
|
||||
},
|
||||
"health_check_max_tokens": 5,
|
||||
"health_check_model": "openai/cheap-model",
|
||||
"health_check_reasoning_effort": "none",
|
||||
}
|
||||
|
||||
updated = _update_litellm_params_for_health_check(model_info, {"model": "openai/dummy"})
|
||||
|
||||
assert updated["max_tokens"] == 5
|
||||
assert updated["model"] == "openai/cheap-model"
|
||||
assert updated["reasoning_effort"] == "none"
|
||||
assert updated["messages"] != model_info["health_check_params"]["messages"]
|
||||
|
||||
|
||||
def test_health_check_params_lose_to_the_audio_speech_voice_knob():
|
||||
"""health_check_voice still wins for audio_speech deployments."""
|
||||
updated = _update_litellm_params_for_health_check(
|
||||
{
|
||||
"mode": "audio_speech",
|
||||
"health_check_params": {"voice": "sage", "response_format": "wav"},
|
||||
"health_check_voice": "shimmer",
|
||||
},
|
||||
{"model": "openai/tts-1"},
|
||||
)
|
||||
|
||||
assert updated["voice"] == "shimmer"
|
||||
assert updated["response_format"] == "wav"
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"bad_value",
|
||||
["mediaSource", ["mediaSource"], 5, True],
|
||||
)
|
||||
def test_health_check_params_ignored_when_not_a_dict(bad_value, caplog):
|
||||
"""A misconfigured health_check_params is skipped with a warning instead of breaking the probe."""
|
||||
with caplog.at_level(logging.WARNING, logger="litellm.proxy.health_check"):
|
||||
updated = _update_litellm_params_for_health_check(
|
||||
{"mode": "chat", "health_check_params": bad_value},
|
||||
{"model": "openai/dummy"},
|
||||
)
|
||||
|
||||
assert updated["model"] == "openai/dummy"
|
||||
assert updated["max_tokens"] == 16
|
||||
assert "health_check_params" in caplog.text
|
||||
|
||||
|
||||
def test_health_check_params_apply_to_non_chat_modes():
|
||||
"""Non-chat probes get health_check_params too, and still no max_tokens."""
|
||||
updated = _update_litellm_params_for_health_check(
|
||||
{"mode": "embedding", "health_check_params": {"dimensions": 8}},
|
||||
{"model": "bedrock/amazon.titan-embed-text-v2:0"},
|
||||
)
|
||||
|
||||
assert updated["dimensions"] == 8
|
||||
assert "max_tokens" not in updated
|
||||
|
||||
|
||||
async def _pegasus_health_check_request_body(
|
||||
model_info: dict[str, object], monkeypatch: pytest.MonkeyPatch
|
||||
) -> dict[str, object]:
|
||||
monkeypatch.setattr(litellm, "disable_aiohttp_transport", True)
|
||||
litellm.in_memory_llm_clients_cache.flush_cache()
|
||||
|
||||
litellm_params = _update_litellm_params_for_health_check(
|
||||
model_info,
|
||||
{
|
||||
"model": "bedrock/us.twelvelabs.pegasus-1-2-v1:0",
|
||||
"aws_access_key_id": "fake-access-key",
|
||||
"aws_secret_access_key": "fake-secret-key",
|
||||
"aws_region_name": "us-east-1",
|
||||
},
|
||||
)
|
||||
|
||||
with respx.mock(assert_all_called=True) as respx_mock:
|
||||
invoke_route = respx_mock.post(
|
||||
host="bedrock-runtime.us-east-1.amazonaws.com",
|
||||
path__regex=r"/model/.+/invoke",
|
||||
).respond(json={"message": "a person walks a dog", "finishReason": "stop"})
|
||||
result = await litellm.ahealth_check(litellm_params, mode="chat")
|
||||
|
||||
assert "error" not in result, result
|
||||
return json.loads(invoke_route.calls.last.request.content)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_health_check_params_reach_the_bedrock_invoke_body(monkeypatch):
|
||||
"""The probe Bedrock actually receives carries mediaSource, which is what unblocks Pegasus."""
|
||||
media_source = {"s3Location": {"uri": "s3://my-bucket/clip.mp4"}}
|
||||
|
||||
body = await _pegasus_health_check_request_body(
|
||||
{"mode": "chat", "health_check_params": {"mediaSource": media_source}}, monkeypatch
|
||||
)
|
||||
|
||||
assert body["mediaSource"] == media_source
|
||||
assert body["maxOutputTokens"] == 16
|
||||
assert body["inputPrompt"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_bedrock_invoke_body_has_no_media_source_without_health_check_params(monkeypatch):
|
||||
"""Negative control: the field only appears because the deployment asked for it."""
|
||||
body = await _pegasus_health_check_request_body({"mode": "chat"}, monkeypatch)
|
||||
|
||||
assert "mediaSource" not in body
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_run_model_health_check_skips_complexity_router_deployment():
|
||||
fake_ahealth_check = AsyncMock(return_value={})
|
||||
|
|
|
|||
|
|
@ -372,6 +372,7 @@ async def test_content__model_encoded_id(harness):
|
|||
async def call_edit(
|
||||
harness: Harness, *, body: Dict[str, Any], headers=None, query=None
|
||||
):
|
||||
harness.read_body.return_value = dict(body)
|
||||
return await endpoints.video_edit(
|
||||
request=FakeRequest(headers=headers, query=query, raw_body=orjson.dumps(body)),
|
||||
fastapi_response=Response(),
|
||||
|
|
@ -428,6 +429,27 @@ async def test_edit__missing_video_object_defaults_to_openai(harness):
|
|||
assert "video" not in data
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_edit__bare_string_video_id_from_form_field(harness):
|
||||
await call_edit(harness, body={"prompt": "brighter", "video": "video_plain"})
|
||||
|
||||
assert harness.processor_data() == {
|
||||
"prompt": "brighter",
|
||||
"video_id": "video_plain",
|
||||
"custom_llm_provider": "openai",
|
||||
}
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_edit__json_string_video_reference_from_form_field(harness):
|
||||
await call_edit(
|
||||
harness,
|
||||
body={"prompt": "brighter", "video": orjson.dumps({"id": "video_plain"}).decode()},
|
||||
)
|
||||
|
||||
assert harness.processor_data()["video_id"] == "video_plain"
|
||||
|
||||
|
||||
# =========================================================================== #
|
||||
# GET /v1/videos - video_list #
|
||||
# =========================================================================== #
|
||||
|
|
@ -471,6 +493,7 @@ async def test_list__provider_from_header(harness):
|
|||
async def call_remix(
|
||||
harness: Harness, video_id: str, *, body, headers=None, query=None
|
||||
):
|
||||
harness.read_body.return_value = dict(body)
|
||||
return await endpoints.video_remix(
|
||||
video_id=video_id,
|
||||
request=FakeRequest(headers=headers, query=query, raw_body=orjson.dumps(body)),
|
||||
|
|
@ -629,6 +652,7 @@ async def test_get_character__plain_id_defaults_openai_no_encode(harness):
|
|||
|
||||
|
||||
async def call_extension(harness: Harness, *, body, headers=None, query=None):
|
||||
harness.read_body.return_value = dict(body)
|
||||
return await endpoints.video_extension(
|
||||
request=FakeRequest(headers=headers, query=query, raw_body=orjson.dumps(body)),
|
||||
fastapi_response=Response(),
|
||||
|
|
|
|||
|
|
@ -1,8 +1,9 @@
|
|||
"""
|
||||
Pure-logic contract tests for litellm/proxy/video_endpoints/utils.py
|
||||
|
||||
Three helpers the video proxy endpoints lean on:
|
||||
Four helpers the video proxy endpoints lean on:
|
||||
- extract_model_from_target_model_names: first model from a comma string / list
|
||||
- video_reference_to_id: normalize a video reference (dict / bare id / JSON string) to an id
|
||||
- get_custom_provider_from_data: provider precedence (top-level > extra_body)
|
||||
- encode_character_id_in_response: re-encode a response id in place
|
||||
|
||||
|
|
@ -20,6 +21,7 @@ from litellm.proxy.video_endpoints.utils import (
|
|||
encode_character_id_in_response,
|
||||
extract_model_from_target_model_names,
|
||||
get_custom_provider_from_data,
|
||||
video_reference_to_id,
|
||||
)
|
||||
from litellm.types.videos.utils import (
|
||||
decode_character_id_with_provider,
|
||||
|
|
@ -53,6 +55,31 @@ def test_extract_model__non_str_non_list_is_none(value):
|
|||
assert extract_model_from_target_model_names(value) is None
|
||||
|
||||
|
||||
# =========================================================================== #
|
||||
# video_reference_to_id
|
||||
# =========================================================================== #
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"video_ref,expected",
|
||||
[
|
||||
({"id": "video_123"}, "video_123"), # dict reference -> its id
|
||||
({"id": ""}, ""), # dict with empty id
|
||||
({}, ""), # dict missing id -> default empty
|
||||
({"other": "x"}, ""), # dict without id key
|
||||
("video_123", "video_123"), # bare id string (not valid JSON) -> itself
|
||||
('{"id": "video_9"}', "video_9"), # JSON-encoded dict -> its id
|
||||
('{"other": 1}', ""), # JSON-encoded dict without id -> empty
|
||||
("[1, 2]", "[1, 2]"), # JSON parses to non-dict -> original string
|
||||
(None, ""), # non-str, non-dict
|
||||
(123, ""), # non-str, non-dict
|
||||
(["video_123"], ""), # list is neither dict nor str
|
||||
],
|
||||
)
|
||||
def test_video_reference_to_id(video_ref, expected):
|
||||
assert video_reference_to_id(video_ref) == expected
|
||||
|
||||
|
||||
# =========================================================================== #
|
||||
# get_custom_provider_from_data
|
||||
# =========================================================================== #
|
||||
|
|
|
|||
|
|
@ -1,6 +1,7 @@
|
|||
import json
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
from typing import Final
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
|
|
@ -41,6 +42,7 @@ from litellm.utils import (
|
|||
get_prompt_cache_min_tokens,
|
||||
is_cached_message,
|
||||
is_prompt_caching_valid_prompt,
|
||||
prompt_token_calculator,
|
||||
)
|
||||
|
||||
# Adds the parent directory to the system path
|
||||
|
|
@ -730,7 +732,9 @@ def validate_model_cost_values(model_data, exceptions=None):
|
|||
"output_cost_per_pixel",
|
||||
"input_cost_per_second",
|
||||
"output_cost_per_second",
|
||||
"output_cost_per_second_480p",
|
||||
"output_cost_per_second_1080p",
|
||||
"output_cost_per_second_4k",
|
||||
"input_cost_per_query",
|
||||
"input_cost_per_request",
|
||||
"input_cost_per_audio_token",
|
||||
|
|
@ -860,7 +864,6 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
|
|||
"input_cost_per_character_above_128k_tokens": {"type": "number"},
|
||||
"input_cost_per_image": {"type": "number"},
|
||||
"input_cost_per_image_above_128k_tokens": {"type": "number"},
|
||||
"input_cost_per_image_token": {"type": "number"},
|
||||
"input_cost_per_video_token": {"type": "number"},
|
||||
"input_cost_per_token_above_200k_tokens": {"type": "number"},
|
||||
"input_cost_per_token_above_256k_tokens": {"type": "number"},
|
||||
|
|
@ -944,7 +947,9 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
|
|||
"output_cost_per_video_token": {"type": "number"},
|
||||
"output_cost_per_pixel": {"type": "number"},
|
||||
"output_cost_per_second": {"type": "number"},
|
||||
"output_cost_per_second_480p": {"type": "number"},
|
||||
"output_cost_per_second_1080p": {"type": "number"},
|
||||
"output_cost_per_second_4k": {"type": "number"},
|
||||
"output_cost_per_token": {"type": "number"},
|
||||
"output_cost_per_token_above_128k_tokens": {"type": "number"},
|
||||
"output_cost_per_token_above_200k_tokens": {"type": "number"},
|
||||
|
|
@ -993,6 +998,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
|
|||
"supports_xhigh_reasoning_effort": {"type": "boolean"},
|
||||
"supports_max_reasoning_effort": {"type": "boolean"},
|
||||
"supports_adaptive_thinking": {"type": "boolean"},
|
||||
"supports_legacy_thinking": {"type": "boolean"},
|
||||
"thinking_always_on": {"type": "boolean"},
|
||||
"supports_mid_conversation_system": {"type": "boolean"},
|
||||
"supports_sampling_params": {"type": "boolean"},
|
||||
|
|
@ -1004,7 +1010,6 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
|
|||
},
|
||||
"bedrock_converse_supports_strict_tools": {"type": "boolean"},
|
||||
"tpm": {"type": "number"},
|
||||
"provider_specific_entry": {"type": "object"},
|
||||
"supported_endpoints": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
|
|
@ -1124,6 +1129,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
|
|||
exceptions = [
|
||||
# Add any model IDs that should be exempt from the cost validation
|
||||
# Example: "expensive-model-id",
|
||||
"runwayml/seedance2", # 4K output is 150 credits/second = $1.50/second
|
||||
]
|
||||
|
||||
is_valid, violations = validate_model_cost_values(actual_json, exceptions)
|
||||
|
|
@ -4973,3 +4979,19 @@ def test_completion_does_not_leak_rust_flag_into_provider_request_body():
|
|||
create_kwargs = mock_client.chat.completions.with_raw_response.create.call_args.kwargs
|
||||
assert "rust" not in create_kwargs
|
||||
assert "rust" not in (create_kwargs.get("extra_body") or {})
|
||||
|
||||
|
||||
def test_prompt_token_calculator_counts_claude_without_the_anthropic_sdk():
|
||||
"""
|
||||
The claude branch used to call the anthropic SDK's `count_tokens`, which the SDK
|
||||
removed, so every claude call raised AttributeError. Counting must work with
|
||||
`anthropic` unimportable.
|
||||
"""
|
||||
messages: Final = [{"role": "user", "content": "the quick brown fox jumps over the lazy dog"}]
|
||||
|
||||
with patch.dict(sys.modules, {"anthropic": None}):
|
||||
claude_tokens = prompt_token_calculator("claude-sonnet-4-5", messages)
|
||||
gpt_tokens = prompt_token_calculator("gpt-4o", messages)
|
||||
|
||||
assert claude_tokens == 9
|
||||
assert gpt_tokens == 9
|
||||
|
|
|
|||
|
|
@ -421,6 +421,117 @@ class TestVideoGeneration:
|
|||
)
|
||||
assert cost == 0.5
|
||||
|
||||
def test_completion_cost_video_custom_pricing_under_litellm_metadata(self):
|
||||
"""Video routes store deployment model_info under litellm_metadata, not metadata.
|
||||
|
||||
Regression for https://github.com/BerriAI/litellm/issues/36483: custom video
|
||||
pricing was silently ignored because completion_cost only read metadata.
|
||||
"""
|
||||
from litellm.cost_calculator import completion_cost
|
||||
|
||||
mock_response = MagicMock()
|
||||
mock_response.usage = {"duration_seconds": 10.0}
|
||||
type(mock_response)._hidden_params = {}
|
||||
|
||||
mock_logging_obj = MagicMock()
|
||||
mock_logging_obj.litellm_params = {
|
||||
"litellm_metadata": {
|
||||
"model_info": {
|
||||
"output_cost_per_video_per_second": 0.18,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
cost = completion_cost(
|
||||
completion_response=mock_response,
|
||||
model="runwayml/seedance2",
|
||||
call_type="create_video",
|
||||
custom_llm_provider="runwayml",
|
||||
custom_pricing=True,
|
||||
litellm_logging_obj=mock_logging_obj,
|
||||
)
|
||||
assert abs(cost - 1.8) < 0.001
|
||||
|
||||
def test_completion_cost_video_uses_provider_reported_cost_without_custom_pricing(self):
|
||||
"""With no custom pricing, the provider's own reported cost wins over a duration estimate."""
|
||||
from litellm.cost_calculator import completion_cost
|
||||
|
||||
mock_response = MagicMock()
|
||||
mock_response.usage = {
|
||||
"duration_seconds": 5.0,
|
||||
"video_resolution": "720p",
|
||||
"provider_reported_cost_usd": 0.31,
|
||||
}
|
||||
type(mock_response)._hidden_params = {}
|
||||
|
||||
cost = completion_cost(
|
||||
completion_response=mock_response,
|
||||
model="runwayml/gen4_turbo",
|
||||
call_type="create_video",
|
||||
custom_llm_provider="runwayml",
|
||||
)
|
||||
assert cost == 0.31
|
||||
|
||||
def test_completion_cost_video_custom_pricing_beats_provider_reported_cost(self):
|
||||
"""Deployment-level custom pricing overrides the provider's reported cost."""
|
||||
from litellm.cost_calculator import completion_cost
|
||||
|
||||
mock_response = MagicMock()
|
||||
mock_response.usage = {
|
||||
"duration_seconds": 10.0,
|
||||
"provider_reported_cost_usd": 0.31,
|
||||
}
|
||||
type(mock_response)._hidden_params = {}
|
||||
|
||||
mock_logging_obj = MagicMock()
|
||||
mock_logging_obj.litellm_params = {
|
||||
"metadata": {
|
||||
"model_info": {
|
||||
"output_cost_per_video_per_second": 0.18,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
cost = completion_cost(
|
||||
completion_response=mock_response,
|
||||
model="runwayml/seedance2",
|
||||
call_type="create_video",
|
||||
custom_llm_provider="runwayml",
|
||||
custom_pricing=True,
|
||||
litellm_logging_obj=mock_logging_obj,
|
||||
)
|
||||
assert abs(cost - 1.8) < 0.001
|
||||
|
||||
def test_completion_cost_video_resolution_tiers_from_cost_map(self, monkeypatch):
|
||||
"""The 480p/1080p/4k tier keys resolve from the shipped runwayml cost map entries."""
|
||||
from litellm.cost_calculator import completion_cost
|
||||
|
||||
local_map_path = os.path.join(
|
||||
os.path.dirname(__file__), "..", "..", "model_prices_and_context_window.json"
|
||||
)
|
||||
with open(local_map_path, "r") as f:
|
||||
monkeypatch.setattr(litellm, "model_cost", json.load(f))
|
||||
|
||||
def cost_for(model: str, resolution: str | None, duration: float) -> float:
|
||||
mock_response = MagicMock()
|
||||
mock_response.usage = {
|
||||
"duration_seconds": duration,
|
||||
**({"video_resolution": resolution} if resolution else {}),
|
||||
}
|
||||
type(mock_response)._hidden_params = {}
|
||||
return completion_cost(
|
||||
completion_response=mock_response,
|
||||
model=model,
|
||||
call_type="create_video",
|
||||
custom_llm_provider="runwayml",
|
||||
)
|
||||
|
||||
assert abs(cost_for("runwayml/seedance2", "4k", 8.0) - 12.0) < 0.001
|
||||
assert abs(cost_for("runwayml/seedance2", "1080p", 8.0) - 3.2) < 0.001
|
||||
assert abs(cost_for("runwayml/seedance2", "720p", 8.0) - 2.88) < 0.001
|
||||
assert abs(cost_for("runwayml/seedance2_5", "480p", 8.0) - 1.6) < 0.001
|
||||
assert abs(cost_for("runwayml/gen4.5", None, 8.0) - 0.96) < 0.001
|
||||
|
||||
def test_video_generation_with_files(self):
|
||||
"""Test video generation with file uploads."""
|
||||
config = OpenAIVideoConfig()
|
||||
|
|
@ -2316,6 +2427,72 @@ def test_edit_and_extension_support_custom_provider_from_extra_body(
|
|||
assert captured_data["custom_llm_provider"] == "vertex_ai"
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"handler_name, path, form",
|
||||
[
|
||||
(
|
||||
"video_edit",
|
||||
"/v1/videos/edits",
|
||||
{"model": "my-video-model", "prompt": "brighter", "video": "video_123"},
|
||||
),
|
||||
(
|
||||
"video_extension",
|
||||
"/v1/videos/extensions",
|
||||
{"model": "my-video-model", "prompt": "continue", "seconds": "4", "video": "video_123"},
|
||||
),
|
||||
],
|
||||
)
|
||||
@pytest.mark.asyncio
|
||||
async def test_edit_and_extension_read_cached_body_after_auth_consumes_stream(
|
||||
handler_name, path, form
|
||||
):
|
||||
from urllib.parse import urlencode
|
||||
|
||||
from fastapi import Response
|
||||
from starlette.requests import Request
|
||||
|
||||
import litellm.proxy.video_endpoints.endpoints as endpoints
|
||||
from litellm.proxy._types import ProxyException, UserAPIKeyAuth
|
||||
from litellm.proxy.common_utils.http_parsing_utils import _read_request_body
|
||||
|
||||
body = urlencode(form).encode()
|
||||
stream = {"sent": False}
|
||||
|
||||
async def receive():
|
||||
if stream["sent"]:
|
||||
return {"type": "http.request", "body": b"", "more_body": False}
|
||||
stream["sent"] = True
|
||||
return {"type": "http.request", "body": body, "more_body": False}
|
||||
|
||||
request = Request(
|
||||
{
|
||||
"type": "http",
|
||||
"method": "POST",
|
||||
"path": path,
|
||||
"headers": [
|
||||
(b"content-type", b"application/x-www-form-urlencoded"),
|
||||
(b"content-length", str(len(body)).encode()),
|
||||
],
|
||||
"query_string": b"",
|
||||
},
|
||||
receive,
|
||||
)
|
||||
|
||||
await _read_request_body(request=request)
|
||||
|
||||
handler = getattr(endpoints, handler_name)
|
||||
with pytest.raises(ProxyException) as exc_info:
|
||||
await handler(
|
||||
request=request,
|
||||
fastapi_response=Response(),
|
||||
user_api_key_dict=UserAPIKeyAuth(api_key="sk-1234"),
|
||||
)
|
||||
|
||||
message = str(exc_info.value)
|
||||
assert "Stream consumed" not in message
|
||||
assert "my-video-model" in message
|
||||
|
||||
|
||||
@pytest.mark.parametrize("endpoint", ["/v1/videos/edits", "/v1/videos/extensions"])
|
||||
def test_edit_and_extension_route_with_encoded_video_ids(
|
||||
video_proxy_test_client, endpoint
|
||||
|
|
|
|||
8
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
8
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
|
|
@ -27624,6 +27624,10 @@ export interface components {
|
|||
output_cost_per_second?: number | null;
|
||||
/** Output Cost Per Second 1080P */
|
||||
output_cost_per_second_1080p?: number | null;
|
||||
/** Output Cost Per Second 480P */
|
||||
output_cost_per_second_480p?: number | null;
|
||||
/** Output Cost Per Second 4K */
|
||||
output_cost_per_second_4k?: number | null;
|
||||
/** Output Cost Per Token */
|
||||
output_cost_per_token?: number | null;
|
||||
/** Output Cost Per Token Above 128K Tokens */
|
||||
|
|
@ -36833,6 +36837,10 @@ export interface components {
|
|||
output_cost_per_second?: number | null;
|
||||
/** Output Cost Per Second 1080P */
|
||||
output_cost_per_second_1080p?: number | null;
|
||||
/** Output Cost Per Second 480P */
|
||||
output_cost_per_second_480p?: number | null;
|
||||
/** Output Cost Per Second 4K */
|
||||
output_cost_per_second_4k?: number | null;
|
||||
/** Output Cost Per Token */
|
||||
output_cost_per_token?: number | null;
|
||||
/** Output Cost Per Token Above 128K Tokens */
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue