fix: apply custom video pricing from deployment model_info (#21923)

* auth_with_role_name add region_name arg for cross-account sts

* update tests to include case with aws_region_name for _auth_with_aws_role

* Only pass region_name to STS client when aws_region_name is set

* Add optional aws_sts_endpoint to _auth_with_aws_role

* Parametrize ambient-credentials test for no opts, region_name, and aws_sts_endpoint

* consistently passing region and endpoint args into explicit credentials irsa

* fix env var leakage

* fix: bedrock openai-compatible imported-model should also have model arn encoded

* fix: custom pricing not applied for /v1/videos endpoint (#21907)

* fix: resolve mypy type errors for video pricing model_info parameter

Use Optional[ModelInfo] instead of Optional[dict] and restructure
cost_info narrowing so mypy can properly track non-None state.

---------

Co-authored-by: An Tang <ta@stripe.com>
Co-authored-by: Sameer Kankute <sameer@berri.ai>
This commit is contained in:
Atharva Jaiswal 2026-02-24 10:14:02 +05:30 committed by Sameer Kankute
parent 0ab380a28d
commit 9565b755a0
3 changed files with 136 additions and 37 deletions

View file

@ -1242,6 +1242,16 @@ def completion_cost( # noqa: PLR0915
)
elif call_type in _VIDEO_CALL_TYPES:
### VIDEO GENERATION COST CALCULATION ###
# Extract custom model_info for deployment-specific pricing
_video_model_info: Optional[ModelInfo] = None
if custom_pricing and litellm_logging_obj is not None:
_litellm_params = getattr(
litellm_logging_obj, "litellm_params", None
)
if _litellm_params is not None:
_metadata = _litellm_params.get("metadata", {}) or {}
_video_model_info = _metadata.get("model_info", None)
usage_obj = getattr(completion_response, "usage", None)
if completion_response is not None and usage_obj:
# Handle both dict and Pydantic Usage object
@ -1262,12 +1272,14 @@ def completion_cost( # noqa: PLR0915
model=model,
duration_seconds=duration_seconds,
custom_llm_provider=custom_llm_provider,
model_info=_video_model_info,
)
# Fallback to default video cost calculation if no duration available
return default_video_cost_calculator(
model=model,
duration_seconds=0.0, # Default to 0 if no duration available
custom_llm_provider=custom_llm_provider,
model_info=_video_model_info,
)
elif call_type in _SPEECH_CALL_TYPES:
prompt_characters = litellm.utils._count_characters(text=prompt)
@ -1892,6 +1904,7 @@ def default_video_cost_calculator(
model: str,
duration_seconds: float,
custom_llm_provider: Optional[str] = None,
model_info: Optional[ModelInfo] = None,
) -> float:
"""
Default video cost calculator for video generation
@ -1900,6 +1913,9 @@ def default_video_cost_calculator(
model (str): Model name
duration_seconds (float): Duration of the generated video in seconds
custom_llm_provider (Optional[str]): Custom LLM provider
model_info (Optional[ModelInfo]): Deployment-level model info containing
custom video pricing. When provided, used before falling back to
the global litellm.model_cost lookup.
Returns:
float: Cost in USD for the video generation
@ -1907,42 +1923,47 @@ def default_video_cost_calculator(
Raises:
Exception: If model pricing not found in cost map
"""
# Build model names for cost lookup
base_model_name = model
model_name_without_custom_llm_provider: Optional[str] = None
if custom_llm_provider and model.startswith(f"{custom_llm_provider}/"):
model_name_without_custom_llm_provider = model.replace(
f"{custom_llm_provider}/", ""
)
base_model_name = (
f"{custom_llm_provider}/{model_name_without_custom_llm_provider}"
)
verbose_logger.debug(f"Looking up cost for video model: {base_model_name}")
model_without_provider = model.split("/")[-1]
# Try model with provider first, fall back to base model name
# Use custom model_info pricing if provided (deployment-specific pricing)
cost_info: Optional[dict] = None
models_to_check: List[Optional[str]] = [
base_model_name,
model,
model_without_provider,
model_name_without_custom_llm_provider,
]
for _model in models_to_check:
if _model is not None and _model in litellm.model_cost:
cost_info = litellm.model_cost[_model]
break
if model_info is not None:
cost_info = dict(model_info)
else:
# Build model names for cost lookup
base_model_name = model
model_name_without_custom_llm_provider: Optional[str] = None
if custom_llm_provider and model.startswith(f"{custom_llm_provider}/"):
model_name_without_custom_llm_provider = model.replace(
f"{custom_llm_provider}/", ""
)
base_model_name = (
f"{custom_llm_provider}/{model_name_without_custom_llm_provider}"
)
verbose_logger.debug(f"Looking up cost for video model: {base_model_name}")
model_without_provider = model.split("/")[-1]
# Try model with provider first, fall back to base model name
models_to_check: List[Optional[str]] = [
base_model_name,
model,
model_without_provider,
model_name_without_custom_llm_provider,
]
for _model in models_to_check:
if _model is not None and _model in litellm.model_cost:
cost_info = litellm.model_cost[_model]
break
# If still not found, try with custom_llm_provider prefix
if cost_info is None and custom_llm_provider:
prefixed_model = f"{custom_llm_provider}/{model}"
if prefixed_model in litellm.model_cost:
cost_info = litellm.model_cost[prefixed_model]
# If still not found, try with custom_llm_provider prefix
if cost_info is None and custom_llm_provider:
prefixed_model = f"{custom_llm_provider}/{model}"
if prefixed_model in litellm.model_cost:
cost_info = litellm.model_cost[prefixed_model]
if cost_info is None:
raise Exception(
f"Model not found in cost map. Tried checking {models_to_check}"
f"Model not found in cost map for model={model}"
)
# Check for video-specific cost per second first

View file

@ -7,7 +7,7 @@ from typing import Literal, Optional, Tuple
from litellm._logging import verbose_logger
from litellm.litellm_core_utils.llm_cost_calc.utils import generic_cost_per_token
from litellm.types.utils import CallTypes, Usage
from litellm.types.utils import CallTypes, ModelInfo, Usage
from litellm.utils import get_model_info
@ -129,7 +129,10 @@ def cost_per_second(
def video_generation_cost(
model: str, duration_seconds: float, custom_llm_provider: Optional[str] = None
model: str,
duration_seconds: float,
custom_llm_provider: Optional[str] = None,
model_info: Optional[ModelInfo] = None,
) -> float:
"""
Calculates the cost for video generation based on duration in seconds.
@ -138,14 +141,18 @@ def video_generation_cost(
- model: str, the model name without provider prefix
- duration_seconds: float, the duration of the generated video in seconds
- custom_llm_provider: str, the custom llm provider
- model_info: Optional[dict], deployment-level model info containing
custom video pricing. When provided, skips the global
get_model_info() lookup so that deployment-specific pricing is used.
Returns:
float - total_cost_in_usd
"""
## GET MODEL INFO
model_info = get_model_info(
model=model, custom_llm_provider=custom_llm_provider or "openai"
)
if model_info is None:
model_info = get_model_info(
model=model, custom_llm_provider=custom_llm_provider or "openai"
)
# Check for video-specific cost per second
video_cost_per_second = model_info.get("output_cost_per_video_per_second")

View file

@ -243,6 +243,77 @@ class TestVideoGeneration:
custom_llm_provider="openai"
)
def test_video_generation_cost_with_custom_model_info(self):
"""Test that custom model_info pricing is applied for video generation.
When a deployment has custom pricing via model_info, it should be used
instead of looking up the global litellm.model_cost map.
Related: https://github.com/BerriAI/litellm/issues/21907
"""
model_info = {
"output_cost_per_video_per_second": 0.05,
}
cost = default_video_cost_calculator(
model="my-custom-video-model",
duration_seconds=10.0,
model_info=model_info,
)
assert cost == 0.5
def test_video_generation_cost_custom_model_info_fallback_to_per_second(self):
"""Test that output_cost_per_second is used as fallback when
output_cost_per_video_per_second is not set in custom model_info.
Related: https://github.com/BerriAI/litellm/issues/21907
"""
model_info = {
"output_cost_per_second": 0.10,
}
cost = default_video_cost_calculator(
model="my-custom-video-model",
duration_seconds=5.0,
model_info=model_info,
)
assert cost == 0.5
def test_video_generation_cost_custom_pricing_through_completion_cost(self):
"""Test that custom video pricing flows through completion_cost via litellm_logging_obj.
This tests the full cost calculation path: completion_cost extracts model_info
from litellm_logging_obj.litellm_params.metadata.model_info and passes it to
the video cost calculator.
Related: https://github.com/BerriAI/litellm/issues/21907
"""
from litellm.cost_calculator import completion_cost
# Create mock response with usage containing duration_seconds
mock_response = MagicMock()
mock_response.usage = MagicMock()
mock_response.usage.duration_seconds = 10.0
type(mock_response)._hidden_params = {}
# Create mock litellm_logging_obj with custom pricing
mock_logging_obj = MagicMock()
mock_logging_obj.litellm_params = {
"metadata": {
"model_info": {
"output_cost_per_video_per_second": 0.05,
}
}
}
cost = completion_cost(
completion_response=mock_response,
model="openai/hunyuanvideo",
call_type="create_video",
custom_llm_provider="openai",
custom_pricing=True,
litellm_logging_obj=mock_logging_obj,
)
assert cost == 0.5
def test_video_generation_with_files(self):
"""Test video generation with file uploads."""
config = OpenAIVideoConfig()