mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
feat(openai): apply regional-processing cost uplift for EU/US data residency
OpenAI charges a 10% uplift on the latest GPT models when requests are served from a regionalized hostname (eu./us.api.openai.com). Infer the region from `api_base`, expose it on `kwargs["litellm_params"]["data_residency"]`, and multiply the computed cost by a per-model `regional_processing_uplift_multiplier_<region>` field. https://claude.ai/code/session_012ebH44s7ohYxjoix5CXzTW
This commit is contained in:
parent
ef36e89638
commit
6f8bb8cd28
13 changed files with 445 additions and 12 deletions
|
|
@ -312,6 +312,10 @@ def cost_per_token( # noqa: PLR0915
|
|||
audio_transcription_file_duration: float = 0.0, # for audio transcription calls - the file time in seconds
|
||||
### SERVICE TIER ###
|
||||
service_tier: Optional[str] = None, # for OpenAI service tier pricing
|
||||
### DATA RESIDENCY ###
|
||||
data_residency: Optional[
|
||||
str
|
||||
] = None, # for OpenAI regional-processing uplift (e.g. "eu", "us")
|
||||
response: Optional[Any] = None,
|
||||
### REQUEST MODEL ###
|
||||
request_model: Optional[str] = None, # original request model for router detection
|
||||
|
|
@ -493,6 +497,7 @@ def cost_per_token( # noqa: PLR0915
|
|||
usage=usage_block,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
service_tier=service_tier,
|
||||
data_residency=data_residency,
|
||||
)
|
||||
|
||||
return prompt_cost, completion_cost
|
||||
|
|
@ -529,6 +534,7 @@ def cost_per_token( # noqa: PLR0915
|
|||
model=model_without_prefix,
|
||||
usage=usage_block,
|
||||
service_tier=service_tier,
|
||||
data_residency=data_residency,
|
||||
)
|
||||
|
||||
return openai_cost_per_second(
|
||||
|
|
@ -579,7 +585,10 @@ def cost_per_token( # noqa: PLR0915
|
|||
)
|
||||
elif custom_llm_provider == "openai":
|
||||
return openai_cost_per_token(
|
||||
model=model, usage=usage_block, service_tier=service_tier
|
||||
model=model,
|
||||
usage=usage_block,
|
||||
service_tier=service_tier,
|
||||
data_residency=data_residency,
|
||||
)
|
||||
elif custom_llm_provider == "databricks":
|
||||
return databricks_cost_per_token(model=model, usage=usage_block)
|
||||
|
|
@ -631,6 +640,7 @@ def cost_per_token( # noqa: PLR0915
|
|||
usage=usage_block,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
service_tier=service_tier,
|
||||
data_residency=data_residency,
|
||||
)
|
||||
|
||||
if (
|
||||
|
|
@ -1117,6 +1127,10 @@ def completion_cost( # noqa: PLR0915
|
|||
litellm_logging_obj: Optional[LitellmLoggingObject] = None,
|
||||
### SERVICE TIER ###
|
||||
service_tier: Optional[str] = None, # for OpenAI service tier pricing
|
||||
### DATA RESIDENCY ###
|
||||
data_residency: Optional[
|
||||
str
|
||||
] = None, # for OpenAI regional-processing uplift (e.g. "eu", "us")
|
||||
) -> float:
|
||||
"""
|
||||
Calculate the cost of a given completion call fot GPT-3.5-turbo, llama2, any litellm supported llm.
|
||||
|
|
@ -1600,6 +1614,7 @@ def completion_cost( # noqa: PLR0915
|
|||
audio_transcription_file_duration=audio_transcription_file_duration,
|
||||
rerank_billed_units=rerank_billed_units,
|
||||
service_tier=service_tier,
|
||||
data_residency=data_residency,
|
||||
response=completion_response,
|
||||
request_model=request_model_for_cost,
|
||||
)
|
||||
|
|
@ -1811,6 +1826,10 @@ def response_cost_calculator(
|
|||
litellm_logging_obj: Optional[LitellmLoggingObject] = None,
|
||||
### SERVICE TIER ###
|
||||
service_tier: Optional[str] = None, # for OpenAI service tier pricing
|
||||
### DATA RESIDENCY ###
|
||||
data_residency: Optional[
|
||||
str
|
||||
] = None, # for OpenAI regional-processing uplift (e.g. "eu", "us")
|
||||
) -> float:
|
||||
"""
|
||||
Returns
|
||||
|
|
@ -1844,6 +1863,7 @@ def response_cost_calculator(
|
|||
router_model_id=router_model_id,
|
||||
litellm_logging_obj=litellm_logging_obj,
|
||||
service_tier=service_tier,
|
||||
data_residency=data_residency,
|
||||
)
|
||||
return response_cost
|
||||
except Exception as e:
|
||||
|
|
|
|||
38
litellm/litellm_core_utils/data_residency.py
Normal file
38
litellm/litellm_core_utils/data_residency.py
Normal file
|
|
@ -0,0 +1,38 @@
|
|||
"""
|
||||
Helpers for resolving OpenAI data-residency (regional processing) from an
|
||||
api_base URL.
|
||||
|
||||
OpenAI enforces hostname-per-region for projects with geography restrictions
|
||||
enabled and rejects requests sent to the wrong host, so the api_base hostname
|
||||
is the authoritative signal of which region a request was processed in.
|
||||
"""
|
||||
|
||||
from typing import Dict, Optional
|
||||
from urllib.parse import urlparse
|
||||
|
||||
# Mapping of OpenAI regional hostnames to the corresponding data-residency
|
||||
# value used by the cost calculator. See
|
||||
# https://developers.openai.com/api/docs/pricing for the regional-processing
|
||||
# uplift these hostnames trigger.
|
||||
_OPENAI_REGIONAL_HOSTS: Dict[str, str] = {
|
||||
"eu.api.openai.com": "eu",
|
||||
"us.api.openai.com": "us",
|
||||
}
|
||||
|
||||
|
||||
def infer_openai_data_residency(api_base: Optional[str]) -> Optional[str]:
|
||||
"""
|
||||
Derive the OpenAI data-residency region from an api_base URL.
|
||||
|
||||
Returns ``"eu"`` for the EU regional host, ``"us"`` for the US regional
|
||||
host, and ``None`` for the default global host (or any non-OpenAI URL).
|
||||
"""
|
||||
if not api_base:
|
||||
return None
|
||||
try:
|
||||
host = urlparse(api_base).hostname
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
if not host:
|
||||
return None
|
||||
return _OPENAI_REGIONAL_HOSTS.get(host.lower())
|
||||
|
|
@ -1,5 +1,7 @@
|
|||
from typing import Optional
|
||||
|
||||
from litellm.litellm_core_utils.data_residency import infer_openai_data_residency
|
||||
|
||||
# Pre-define optional kwargs keys as frozenset for O(1) lookups
|
||||
# These are extracted from kwargs only if present, avoiding unnecessary .get() calls
|
||||
_OPTIONAL_KWARGS_KEYS = frozenset(
|
||||
|
|
@ -103,6 +105,15 @@ def get_litellm_params(
|
|||
if litellm_trace_id is None:
|
||||
litellm_trace_id = _meta.get("trace_id") or _meta.get("session_id")
|
||||
|
||||
# Derive data_residency from an OpenAI regional api_base (eu./us.api.openai.com)
|
||||
# so custom callbacks can read kwargs["litellm_params"]["data_residency"]
|
||||
# without having to parse the URL.
|
||||
data_residency: Optional[str] = (
|
||||
infer_openai_data_residency(api_base)
|
||||
if custom_llm_provider == "openai" or custom_llm_provider is None
|
||||
else None
|
||||
)
|
||||
|
||||
# Build base dict with explicit parameters (always included)
|
||||
litellm_params = {
|
||||
"acompletion": acompletion,
|
||||
|
|
@ -112,6 +123,7 @@ def get_litellm_params(
|
|||
"verbose": verbose,
|
||||
"custom_llm_provider": custom_llm_provider,
|
||||
"api_base": api_base,
|
||||
"data_residency": data_residency,
|
||||
"litellm_call_id": litellm_call_id,
|
||||
"model_alias_map": model_alias_map,
|
||||
"completion_call_id": completion_call_id,
|
||||
|
|
|
|||
|
|
@ -1546,6 +1546,11 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
if self.optional_params
|
||||
else None
|
||||
),
|
||||
"data_residency": (
|
||||
self.litellm_params.get("data_residency")
|
||||
if hasattr(self, "litellm_params") and self.litellm_params
|
||||
else None
|
||||
),
|
||||
}
|
||||
except Exception as e: # error creating kwargs for cost calculation
|
||||
debug_info = StandardLoggingModelCostFailureDebugInformation(
|
||||
|
|
|
|||
|
|
@ -9,6 +9,7 @@ from litellm.types.utils import (
|
|||
CacheCreationTokenDetails,
|
||||
CallTypes,
|
||||
CompletionTokensDetailsWrapper,
|
||||
DataResidency,
|
||||
ImageResponse,
|
||||
ModelInfo,
|
||||
PassthroughCallTypes,
|
||||
|
|
@ -617,11 +618,46 @@ def _calculate_input_cost(
|
|||
return prompt_cost
|
||||
|
||||
|
||||
def _get_regional_uplift_multiplier(
|
||||
model_info: ModelInfo, data_residency: Optional[str]
|
||||
) -> float:
|
||||
"""
|
||||
Resolve the per-model regional-processing uplift multiplier for a given
|
||||
data-residency region.
|
||||
|
||||
OpenAI applies a flat percentage uplift (e.g. +10%) on all token costs for
|
||||
requests served from a regionalized hostname (eu./us.api.openai.com). The
|
||||
multiplier is stored on the model entry as
|
||||
``regional_processing_uplift_multiplier_<region>`` (e.g. 1.10).
|
||||
|
||||
Returns 1.0 (no uplift) when ``data_residency`` is ``None`` or when the
|
||||
model has no multiplier configured for the given region.
|
||||
"""
|
||||
if data_residency is None:
|
||||
return 1.0
|
||||
residency = data_residency.lower()
|
||||
if residency not in {r.value for r in DataResidency}:
|
||||
return 1.0
|
||||
multiplier = model_info.get(f"regional_processing_uplift_multiplier_{residency}")
|
||||
if multiplier is None:
|
||||
return 1.0
|
||||
try:
|
||||
return float(cast(float, multiplier))
|
||||
except (TypeError, ValueError):
|
||||
verbose_logger.exception(
|
||||
"Invalid regional_processing_uplift_multiplier_%s for model; "
|
||||
"defaulting to 1.0",
|
||||
residency,
|
||||
)
|
||||
return 1.0
|
||||
|
||||
|
||||
def generic_cost_per_token( # noqa: PLR0915
|
||||
model: str,
|
||||
usage: Usage,
|
||||
custom_llm_provider: str,
|
||||
service_tier: Optional[str] = None,
|
||||
data_residency: Optional[str] = None,
|
||||
) -> Tuple[float, float]:
|
||||
"""
|
||||
Calculates the cost per token for a given model, prompt tokens, and completion tokens.
|
||||
|
|
@ -631,6 +667,8 @@ def generic_cost_per_token( # noqa: PLR0915
|
|||
Input:
|
||||
- model: str, the model name without provider prefix
|
||||
- usage: LiteLLM Usage block, containing anthropic caching information
|
||||
- data_residency: optional OpenAI data-residency region (e.g. "eu", "us"),
|
||||
used to apply the per-model regional-processing uplift multiplier.
|
||||
|
||||
Returns:
|
||||
Tuple[float, float] - prompt_cost_in_usd, completion_cost_in_usd
|
||||
|
|
@ -781,6 +819,14 @@ def generic_cost_per_token( # noqa: PLR0915
|
|||
)
|
||||
completion_cost += float(image_tokens) * _output_cost_per_image_token
|
||||
|
||||
## REGIONAL DATA-RESIDENCY UPLIFT
|
||||
# Applied as a flat multiplier across all token costs for the request
|
||||
# when the upstream is a regionalized OpenAI host (eu./us.api.openai.com).
|
||||
uplift = _get_regional_uplift_multiplier(model_info, data_residency)
|
||||
if uplift != 1.0:
|
||||
prompt_cost *= uplift
|
||||
completion_cost *= uplift
|
||||
|
||||
return prompt_cost, completion_cost
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -19,7 +19,10 @@ def cost_router(call_type: CallTypes) -> Literal["cost_per_token", "cost_per_sec
|
|||
|
||||
|
||||
def cost_per_token(
|
||||
model: str, usage: Usage, service_tier: Optional[str] = None
|
||||
model: str,
|
||||
usage: Usage,
|
||||
service_tier: Optional[str] = None,
|
||||
data_residency: Optional[str] = None,
|
||||
) -> Tuple[float, float]:
|
||||
"""
|
||||
Calculates the cost per token for a given model, prompt tokens, and completion tokens.
|
||||
|
|
@ -27,6 +30,9 @@ def cost_per_token(
|
|||
Input:
|
||||
- model: str, the model name without provider prefix
|
||||
- usage: LiteLLM Usage block, containing anthropic caching information
|
||||
- data_residency: optional OpenAI data-residency region (e.g. "eu", "us"),
|
||||
inferred from api_base. Applies the model's regional-processing
|
||||
uplift multiplier when set.
|
||||
|
||||
Returns:
|
||||
Tuple[float, float] - prompt_cost_in_usd, completion_cost_in_usd
|
||||
|
|
@ -37,6 +43,7 @@ def cost_per_token(
|
|||
usage=usage,
|
||||
custom_llm_provider="openai",
|
||||
service_tier=service_tier,
|
||||
data_residency=data_residency,
|
||||
)
|
||||
# ### Non-cached text tokens
|
||||
# non_cached_text_tokens = usage.prompt_tokens
|
||||
|
|
|
|||
|
|
@ -1011,6 +1011,7 @@
|
|||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true,
|
||||
"supports_output_config": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
|
|
@ -1041,6 +1042,7 @@
|
|||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true,
|
||||
"supports_output_config": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
|
|
@ -1071,6 +1073,7 @@
|
|||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true,
|
||||
"supports_output_config": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
|
|
@ -1100,6 +1103,7 @@
|
|||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true,
|
||||
"supports_output_config": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
|
|
@ -1129,6 +1133,7 @@
|
|||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true,
|
||||
"supports_output_config": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
|
|
@ -1328,6 +1333,7 @@
|
|||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true,
|
||||
"supports_output_config": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"global.anthropic.claude-sonnet-4-6": {
|
||||
|
|
@ -1358,6 +1364,7 @@
|
|||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true,
|
||||
"supports_output_config": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"us.anthropic.claude-sonnet-4-6": {
|
||||
|
|
@ -1388,6 +1395,7 @@
|
|||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true,
|
||||
"supports_output_config": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"eu.anthropic.claude-sonnet-4-6": {
|
||||
|
|
@ -1417,6 +1425,7 @@
|
|||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true,
|
||||
"supports_output_config": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"au.anthropic.claude-sonnet-4-6": {
|
||||
|
|
@ -1446,6 +1455,7 @@
|
|||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true,
|
||||
"supports_output_config": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"jp.anthropic.claude-sonnet-4-6": {
|
||||
|
|
@ -1475,6 +1485,7 @@
|
|||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_native_structured_output": true,
|
||||
"supports_output_config": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"anthropic.claude-sonnet-4-20250514-v1:0": {
|
||||
|
|
@ -1996,6 +2007,7 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 159,
|
||||
"supports_output_config": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
|
|
@ -2093,6 +2105,7 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_output_config": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"azure/computer-use-preview": {
|
||||
|
|
@ -9643,6 +9656,7 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_output_config": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"claude-sonnet-4-5-20250929-v1:0": {
|
||||
|
|
@ -9840,6 +9854,7 @@
|
|||
"us": 1.1,
|
||||
"fast": 6.0
|
||||
},
|
||||
"supports_output_config": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
|
|
@ -9875,7 +9890,8 @@
|
|||
"fast": 6.0
|
||||
},
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
"supports_minimal_reasoning_effort": true,
|
||||
"supports_output_config": true
|
||||
},
|
||||
"claude-opus-4-7": {
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
|
|
@ -9910,7 +9926,8 @@
|
|||
"us": 1.1,
|
||||
"fast": 6.0
|
||||
},
|
||||
"supports_minimal_reasoning_effort": true
|
||||
"supports_minimal_reasoning_effort": true,
|
||||
"supports_output_config": true
|
||||
},
|
||||
"claude-opus-4-7-20260416": {
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
|
|
@ -9945,7 +9962,8 @@
|
|||
"us": 1.1,
|
||||
"fast": 6.0
|
||||
},
|
||||
"supports_minimal_reasoning_effort": true
|
||||
"supports_minimal_reasoning_effort": true,
|
||||
"supports_output_config": true
|
||||
},
|
||||
"claude-sonnet-4-20250514": {
|
||||
"deprecation_date": "2026-05-14",
|
||||
|
|
@ -14947,7 +14965,7 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_reasoning_token": 1.5e-06,
|
||||
"output_cost_per_token": 1.5e-06,
|
||||
"source": "https://cloud.google.com/vertex-ai/generative-ai/pricing#gemini-models",
|
||||
"source": "https://ai.google.dev/gemini-api/docs/models",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/completions",
|
||||
|
|
@ -19003,6 +19021,8 @@
|
|||
"output_cost_per_token": 8e-06,
|
||||
"output_cost_per_token_batches": 4e-06,
|
||||
"output_cost_per_token_priority": 1.4e-05,
|
||||
"regional_processing_uplift_multiplier_eu": 1.10,
|
||||
"regional_processing_uplift_multiplier_us": 1.10,
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/batch",
|
||||
|
|
@ -19076,6 +19096,8 @@
|
|||
"output_cost_per_token": 1.6e-06,
|
||||
"output_cost_per_token_batches": 8e-07,
|
||||
"output_cost_per_token_priority": 2.8e-06,
|
||||
"regional_processing_uplift_multiplier_eu": 1.10,
|
||||
"regional_processing_uplift_multiplier_us": 1.10,
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/batch",
|
||||
|
|
@ -19149,6 +19171,8 @@
|
|||
"output_cost_per_token": 4e-07,
|
||||
"output_cost_per_token_batches": 2e-07,
|
||||
"output_cost_per_token_priority": 8e-07,
|
||||
"regional_processing_uplift_multiplier_eu": 1.10,
|
||||
"regional_processing_uplift_multiplier_us": 1.10,
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/batch",
|
||||
|
|
@ -19220,6 +19244,8 @@
|
|||
"output_cost_per_token": 1e-05,
|
||||
"output_cost_per_token_batches": 5e-06,
|
||||
"output_cost_per_token_priority": 1.7e-05,
|
||||
"regional_processing_uplift_multiplier_eu": 1.10,
|
||||
"regional_processing_uplift_multiplier_us": 1.10,
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_pdf_input": true,
|
||||
|
|
@ -19261,6 +19287,8 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 1e-05,
|
||||
"output_cost_per_token_batches": 5e-06,
|
||||
"regional_processing_uplift_multiplier_eu": 1.10,
|
||||
"regional_processing_uplift_multiplier_us": 1.10,
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_pdf_input": true,
|
||||
|
|
@ -19282,6 +19310,8 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 1e-05,
|
||||
"output_cost_per_token_batches": 5e-06,
|
||||
"regional_processing_uplift_multiplier_eu": 1.10,
|
||||
"regional_processing_uplift_multiplier_us": 1.10,
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_pdf_input": true,
|
||||
|
|
@ -19570,6 +19600,8 @@
|
|||
"output_cost_per_token": 6e-07,
|
||||
"output_cost_per_token_batches": 3e-07,
|
||||
"output_cost_per_token_priority": 1e-06,
|
||||
"regional_processing_uplift_multiplier_eu": 1.10,
|
||||
"regional_processing_uplift_multiplier_us": 1.10,
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_pdf_input": true,
|
||||
|
|
@ -20273,6 +20305,8 @@
|
|||
"output_cost_per_token": 1e-05,
|
||||
"output_cost_per_token_flex": 5e-06,
|
||||
"output_cost_per_token_priority": 2e-05,
|
||||
"regional_processing_uplift_multiplier_eu": 1.10,
|
||||
"regional_processing_uplift_multiplier_us": 1.10,
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/batch",
|
||||
|
|
@ -21195,6 +21229,8 @@
|
|||
"mode": "responses",
|
||||
"output_cost_per_token": 0.00012,
|
||||
"output_cost_per_token_batches": 6e-05,
|
||||
"regional_processing_uplift_multiplier_eu": 1.10,
|
||||
"regional_processing_uplift_multiplier_us": 1.10,
|
||||
"supported_endpoints": [
|
||||
"/v1/batch",
|
||||
"/v1/responses"
|
||||
|
|
@ -21601,6 +21637,8 @@
|
|||
"output_cost_per_token": 2e-06,
|
||||
"output_cost_per_token_flex": 1e-06,
|
||||
"output_cost_per_token_priority": 3.6e-06,
|
||||
"regional_processing_uplift_multiplier_eu": 1.10,
|
||||
"regional_processing_uplift_multiplier_us": 1.10,
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/batch",
|
||||
|
|
@ -21682,6 +21720,8 @@
|
|||
"max_input_tokens": 272000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"regional_processing_uplift_multiplier_eu": 1.10,
|
||||
"regional_processing_uplift_multiplier_us": 1.10,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 4e-07,
|
||||
"output_cost_per_token_flex": 2e-07,
|
||||
|
|
@ -28187,10 +28227,10 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"openrouter/xiaomi/mimo-v2-flash": {
|
||||
"input_cost_per_token": 9e-08,
|
||||
"output_cost_per_token": 2.9e-07,
|
||||
"input_cost_per_token": 1e-07,
|
||||
"output_cost_per_token": 3e-07,
|
||||
"cache_creation_input_token_cost": 0.0,
|
||||
"cache_read_input_token_cost": 0.0,
|
||||
"cache_read_input_token_cost": 1e-08,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 16384,
|
||||
|
|
@ -28200,7 +28240,43 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_vision": false,
|
||||
"supports_prompt_caching": false
|
||||
"supports_prompt_caching": true
|
||||
},
|
||||
"openrouter/xiaomi/mimo-v2.5-pro": {
|
||||
"input_cost_per_token": 1e-06,
|
||||
"output_cost_per_token": 3e-06,
|
||||
"cache_creation_input_token_cost": 0.0,
|
||||
"cache_read_input_token_cost": 2e-07,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 16384,
|
||||
"max_tokens": 16384,
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_vision": false,
|
||||
"supports_response_schema": true,
|
||||
"supports_prompt_caching": true
|
||||
},
|
||||
"openrouter/xiaomi/mimo-v2.5": {
|
||||
"input_cost_per_token": 4e-07,
|
||||
"output_cost_per_token": 2e-06,
|
||||
"cache_creation_input_token_cost": 0.0,
|
||||
"cache_read_input_token_cost": 8e-08,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 131072,
|
||||
"max_tokens": 131072,
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_vision": true,
|
||||
"supports_audio_input": true,
|
||||
"supports_video_input": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_prompt_caching": true
|
||||
},
|
||||
"openrouter/z-ai/glm-4.7": {
|
||||
"input_cost_per_token": 4e-07,
|
||||
|
|
@ -28931,14 +29007,16 @@
|
|||
"mode": "responses",
|
||||
"supports_web_search": true,
|
||||
"supports_reasoning": false,
|
||||
"supports_function_calling": true
|
||||
"supports_function_calling": true,
|
||||
"supports_output_config": true
|
||||
},
|
||||
"perplexity/anthropic/claude-opus-4-7": {
|
||||
"litellm_provider": "perplexity",
|
||||
"mode": "responses",
|
||||
"supports_web_search": true,
|
||||
"supports_reasoning": false,
|
||||
"supports_function_calling": true
|
||||
"supports_function_calling": true,
|
||||
"supports_output_config": true
|
||||
},
|
||||
"perplexity/anthropic/claude-opus-4-5": {
|
||||
"litellm_provider": "perplexity",
|
||||
|
|
@ -33349,6 +33427,7 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_output_config": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
|
|
@ -33377,6 +33456,7 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"tool_use_system_prompt_tokens": 346,
|
||||
"supports_output_config": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
|
|
@ -33490,6 +33570,7 @@
|
|||
"search_context_size_low": 0.01,
|
||||
"search_context_size_medium": 0.01
|
||||
},
|
||||
"supports_output_config": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"vertex_ai/claude-sonnet-4-5@20250929": {
|
||||
|
|
@ -40602,6 +40683,7 @@
|
|||
"search_context_size_low": 0.01,
|
||||
"search_context_size_medium": 0.01
|
||||
},
|
||||
"supports_output_config": true,
|
||||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"duckduckgo/search": {
|
||||
|
|
|
|||
|
|
@ -219,6 +219,12 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False):
|
|||
output_cost_per_token_priority: Optional[
|
||||
float
|
||||
] # OpenAI priority service tier pricing
|
||||
regional_processing_uplift_multiplier_eu: Optional[
|
||||
float
|
||||
] # OpenAI EU data-residency uplift multiplier applied to all token costs (e.g. 1.10 = +10%)
|
||||
regional_processing_uplift_multiplier_us: Optional[
|
||||
float
|
||||
] # OpenAI US data-residency uplift multiplier applied to all token costs (e.g. 1.10 = +10%)
|
||||
output_cost_per_character: Optional[float] # only for vertex ai models
|
||||
output_cost_per_audio_token: Optional[float]
|
||||
output_cost_per_token_above_128k_tokens: Optional[
|
||||
|
|
@ -3601,6 +3607,20 @@ class ServiceTier(Enum):
|
|||
PRIORITY = "priority"
|
||||
|
||||
|
||||
class DataResidency(Enum):
|
||||
"""
|
||||
OpenAI data-residency / regional-processing regions.
|
||||
|
||||
Inferred from the OpenAI api_base host (eu.api.openai.com -> EU,
|
||||
us.api.openai.com -> US). Used to apply the regional-processing
|
||||
cost uplift (see ``regional_processing_uplift_multiplier_<region>``
|
||||
on ModelInfo).
|
||||
"""
|
||||
|
||||
US = "us"
|
||||
EU = "eu"
|
||||
|
||||
|
||||
LLMResponseTypes = Union[
|
||||
ModelResponse,
|
||||
EmbeddingResponse,
|
||||
|
|
|
|||
|
|
@ -5902,6 +5902,12 @@ def _get_model_info_helper( # noqa: PLR0915
|
|||
output_cost_per_token_priority=_model_info.get(
|
||||
"output_cost_per_token_priority", None
|
||||
),
|
||||
regional_processing_uplift_multiplier_eu=_model_info.get(
|
||||
"regional_processing_uplift_multiplier_eu", None
|
||||
),
|
||||
regional_processing_uplift_multiplier_us=_model_info.get(
|
||||
"regional_processing_uplift_multiplier_us", None
|
||||
),
|
||||
output_cost_per_audio_token=_model_info.get(
|
||||
"output_cost_per_audio_token", None
|
||||
),
|
||||
|
|
|
|||
|
|
@ -19021,6 +19021,8 @@
|
|||
"output_cost_per_token": 8e-06,
|
||||
"output_cost_per_token_batches": 4e-06,
|
||||
"output_cost_per_token_priority": 1.4e-05,
|
||||
"regional_processing_uplift_multiplier_eu": 1.10,
|
||||
"regional_processing_uplift_multiplier_us": 1.10,
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/batch",
|
||||
|
|
@ -19094,6 +19096,8 @@
|
|||
"output_cost_per_token": 1.6e-06,
|
||||
"output_cost_per_token_batches": 8e-07,
|
||||
"output_cost_per_token_priority": 2.8e-06,
|
||||
"regional_processing_uplift_multiplier_eu": 1.10,
|
||||
"regional_processing_uplift_multiplier_us": 1.10,
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/batch",
|
||||
|
|
@ -19167,6 +19171,8 @@
|
|||
"output_cost_per_token": 4e-07,
|
||||
"output_cost_per_token_batches": 2e-07,
|
||||
"output_cost_per_token_priority": 8e-07,
|
||||
"regional_processing_uplift_multiplier_eu": 1.10,
|
||||
"regional_processing_uplift_multiplier_us": 1.10,
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/batch",
|
||||
|
|
@ -19238,6 +19244,8 @@
|
|||
"output_cost_per_token": 1e-05,
|
||||
"output_cost_per_token_batches": 5e-06,
|
||||
"output_cost_per_token_priority": 1.7e-05,
|
||||
"regional_processing_uplift_multiplier_eu": 1.10,
|
||||
"regional_processing_uplift_multiplier_us": 1.10,
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_pdf_input": true,
|
||||
|
|
@ -19279,6 +19287,8 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 1e-05,
|
||||
"output_cost_per_token_batches": 5e-06,
|
||||
"regional_processing_uplift_multiplier_eu": 1.10,
|
||||
"regional_processing_uplift_multiplier_us": 1.10,
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_pdf_input": true,
|
||||
|
|
@ -19300,6 +19310,8 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 1e-05,
|
||||
"output_cost_per_token_batches": 5e-06,
|
||||
"regional_processing_uplift_multiplier_eu": 1.10,
|
||||
"regional_processing_uplift_multiplier_us": 1.10,
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_pdf_input": true,
|
||||
|
|
@ -19588,6 +19600,8 @@
|
|||
"output_cost_per_token": 6e-07,
|
||||
"output_cost_per_token_batches": 3e-07,
|
||||
"output_cost_per_token_priority": 1e-06,
|
||||
"regional_processing_uplift_multiplier_eu": 1.10,
|
||||
"regional_processing_uplift_multiplier_us": 1.10,
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_pdf_input": true,
|
||||
|
|
@ -20291,6 +20305,8 @@
|
|||
"output_cost_per_token": 1e-05,
|
||||
"output_cost_per_token_flex": 5e-06,
|
||||
"output_cost_per_token_priority": 2e-05,
|
||||
"regional_processing_uplift_multiplier_eu": 1.10,
|
||||
"regional_processing_uplift_multiplier_us": 1.10,
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/batch",
|
||||
|
|
@ -21213,6 +21229,8 @@
|
|||
"mode": "responses",
|
||||
"output_cost_per_token": 0.00012,
|
||||
"output_cost_per_token_batches": 6e-05,
|
||||
"regional_processing_uplift_multiplier_eu": 1.10,
|
||||
"regional_processing_uplift_multiplier_us": 1.10,
|
||||
"supported_endpoints": [
|
||||
"/v1/batch",
|
||||
"/v1/responses"
|
||||
|
|
@ -21619,6 +21637,8 @@
|
|||
"output_cost_per_token": 2e-06,
|
||||
"output_cost_per_token_flex": 1e-06,
|
||||
"output_cost_per_token_priority": 3.6e-06,
|
||||
"regional_processing_uplift_multiplier_eu": 1.10,
|
||||
"regional_processing_uplift_multiplier_us": 1.10,
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/batch",
|
||||
|
|
@ -21700,6 +21720,8 @@
|
|||
"max_input_tokens": 272000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"regional_processing_uplift_multiplier_eu": 1.10,
|
||||
"regional_processing_uplift_multiplier_us": 1.10,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 4e-07,
|
||||
"output_cost_per_token_flex": 2e-07,
|
||||
|
|
|
|||
|
|
@ -1418,3 +1418,115 @@ def test_image_count_prevents_text_tokens_fallback():
|
|||
f"got {prompt_cost}. text_tokens fallback may be double-charging."
|
||||
)
|
||||
assert completion_cost == 0.0
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Data-residency (OpenAI regional processing) tests
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def _local_model_cost_map():
|
||||
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
|
||||
litellm.model_cost = litellm.get_model_cost_map(url="")
|
||||
yield
|
||||
os.environ.pop("LITELLM_LOCAL_MODEL_COST_MAP", None)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("data_residency", ["eu", "us"])
|
||||
def test_data_residency_applies_uplift(data_residency, _local_model_cost_map):
|
||||
"""gpt-5 should apply the regional processing uplift multiplier when
|
||||
data_residency is set."""
|
||||
from litellm.types.utils import Usage
|
||||
|
||||
usage = Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500)
|
||||
|
||||
base = generic_cost_per_token(
|
||||
model="gpt-5",
|
||||
usage=usage,
|
||||
custom_llm_provider="openai",
|
||||
)
|
||||
regional = generic_cost_per_token(
|
||||
model="gpt-5",
|
||||
usage=usage,
|
||||
custom_llm_provider="openai",
|
||||
data_residency=data_residency,
|
||||
)
|
||||
|
||||
base_total = base[0] + base[1]
|
||||
regional_total = regional[0] + regional[1]
|
||||
|
||||
assert base_total > 0
|
||||
assert regional_total == pytest.approx(base_total * 1.10, rel=1e-9)
|
||||
assert regional[0] == pytest.approx(base[0] * 1.10, rel=1e-9)
|
||||
assert regional[1] == pytest.approx(base[1] * 1.10, rel=1e-9)
|
||||
|
||||
|
||||
def test_data_residency_no_uplift_for_unmarked_model(_local_model_cost_map):
|
||||
"""A model without a regional_processing_uplift_multiplier_* entry should
|
||||
fall back to base pricing, not error."""
|
||||
from litellm.types.utils import Usage
|
||||
|
||||
usage = Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500)
|
||||
|
||||
base = generic_cost_per_token(
|
||||
model="gpt-3.5-turbo",
|
||||
usage=usage,
|
||||
custom_llm_provider="openai",
|
||||
)
|
||||
with_residency = generic_cost_per_token(
|
||||
model="gpt-3.5-turbo",
|
||||
usage=usage,
|
||||
custom_llm_provider="openai",
|
||||
data_residency="eu",
|
||||
)
|
||||
|
||||
assert base == with_residency
|
||||
|
||||
|
||||
def test_data_residency_none_no_uplift(_local_model_cost_map):
|
||||
"""data_residency=None should be a no-op even for models with a multiplier."""
|
||||
from litellm.types.utils import Usage
|
||||
|
||||
usage = Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500)
|
||||
|
||||
base = generic_cost_per_token(
|
||||
model="gpt-5",
|
||||
usage=usage,
|
||||
custom_llm_provider="openai",
|
||||
)
|
||||
explicit_none = generic_cost_per_token(
|
||||
model="gpt-5",
|
||||
usage=usage,
|
||||
custom_llm_provider="openai",
|
||||
data_residency=None,
|
||||
)
|
||||
|
||||
assert base == explicit_none
|
||||
|
||||
|
||||
def test_data_residency_composes_with_service_tier(_local_model_cost_map):
|
||||
"""The uplift multiplies the priority-tier cost, not the standard one."""
|
||||
from litellm.types.utils import Usage
|
||||
|
||||
usage = Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500)
|
||||
|
||||
priority_base = generic_cost_per_token(
|
||||
model="gpt-5",
|
||||
usage=usage,
|
||||
custom_llm_provider="openai",
|
||||
service_tier="priority",
|
||||
)
|
||||
priority_eu = generic_cost_per_token(
|
||||
model="gpt-5",
|
||||
usage=usage,
|
||||
custom_llm_provider="openai",
|
||||
service_tier="priority",
|
||||
data_residency="eu",
|
||||
)
|
||||
|
||||
priority_base_total = priority_base[0] + priority_base[1]
|
||||
priority_eu_total = priority_eu[0] + priority_eu[1]
|
||||
|
||||
assert priority_base_total > 0
|
||||
assert priority_eu_total == pytest.approx(priority_base_total * 1.10, rel=1e-9)
|
||||
|
|
|
|||
26
tests/test_litellm/litellm_core_utils/test_data_residency.py
Normal file
26
tests/test_litellm/litellm_core_utils/test_data_residency.py
Normal file
|
|
@ -0,0 +1,26 @@
|
|||
"""Tests for the OpenAI data-residency inference helper."""
|
||||
|
||||
import pytest
|
||||
|
||||
from litellm.litellm_core_utils.data_residency import infer_openai_data_residency
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"api_base, expected",
|
||||
[
|
||||
("https://eu.api.openai.com/v1", "eu"),
|
||||
("https://eu.api.openai.com", "eu"),
|
||||
("https://us.api.openai.com/v1", "us"),
|
||||
("https://us.api.openai.com", "us"),
|
||||
("https://EU.api.openai.com/v1", "eu"),
|
||||
("https://api.openai.com/v1", None),
|
||||
("https://api.openai.com", None),
|
||||
("https://example.com/v1", None),
|
||||
("https://my-azure-endpoint.openai.azure.com/openai/deployments/foo", None),
|
||||
("", None),
|
||||
(None, None),
|
||||
("not a url", None),
|
||||
],
|
||||
)
|
||||
def test_infer_openai_data_residency(api_base, expected):
|
||||
assert infer_openai_data_residency(api_base) == expected
|
||||
|
|
@ -125,3 +125,40 @@ class TestGetLitellmParamsExplicitFields:
|
|||
def test_no_log_from_explicit_param(self):
|
||||
result = get_litellm_params(no_log=True)
|
||||
assert result["no-log"] is True
|
||||
|
||||
|
||||
class TestGetLitellmParamsDataResidency:
|
||||
"""Verify that data_residency is inferred from OpenAI regional api_base."""
|
||||
|
||||
def test_eu_host_resolves_to_eu(self):
|
||||
result = get_litellm_params(
|
||||
custom_llm_provider="openai",
|
||||
api_base="https://eu.api.openai.com/v1",
|
||||
)
|
||||
assert result["data_residency"] == "eu"
|
||||
|
||||
def test_us_host_resolves_to_us(self):
|
||||
result = get_litellm_params(
|
||||
custom_llm_provider="openai",
|
||||
api_base="https://us.api.openai.com/v1",
|
||||
)
|
||||
assert result["data_residency"] == "us"
|
||||
|
||||
def test_global_host_resolves_to_none(self):
|
||||
result = get_litellm_params(
|
||||
custom_llm_provider="openai",
|
||||
api_base="https://api.openai.com/v1",
|
||||
)
|
||||
assert result["data_residency"] is None
|
||||
|
||||
def test_no_api_base_is_none(self):
|
||||
result = get_litellm_params(custom_llm_provider="openai")
|
||||
assert result["data_residency"] is None
|
||||
|
||||
def test_non_openai_provider_does_not_resolve(self):
|
||||
"""Regional OpenAI host doesn't apply to other providers."""
|
||||
result = get_litellm_params(
|
||||
custom_llm_provider="anthropic",
|
||||
api_base="https://eu.api.openai.com/v1",
|
||||
)
|
||||
assert result["data_residency"] is None
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue