mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
fix(router): bill service tiers at catalog rates for custom-priced deployments (#43890)
* fix(router): inherit catalog service-tier rates for custom-priced deployments Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(cost): apply tier-suffixed long-context rates when only tier thresholds are set Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(router): cover canonical cost-map backend model resolution Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(router): use descriptive names for service-tier pricing fixtures Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: kerry <kerry@berri.ai> Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
6fd9334751
commit
c42d06fb80
5 changed files with 708 additions and 4 deletions
|
|
@ -76,6 +76,9 @@ _SERVICE_TIER_TO_COST_KEY_SUFFIX: Final[Mapping[str, str]] = MappingProxyType(
|
|||
ServiceTier.ULTRAFAST.value: ServiceTier.ULTRAFAST.value,
|
||||
}
|
||||
)
|
||||
SERVICE_TIER_COST_KEY_SUFFIXES: Final[tuple[str, ...]] = tuple(
|
||||
sorted(frozenset(f"_{suffix}" for suffix in _SERVICE_TIER_TO_COST_KEY_SUFFIX.values()))
|
||||
)
|
||||
|
||||
_INCLUSIVE_THRESHOLD_PROVIDERS: Final = frozenset({"xai"})
|
||||
_BATCH_KEY_SUFFIX: Final = "_batches"
|
||||
|
|
@ -663,13 +666,15 @@ def _get_token_base_cost(
|
|||
|
||||
## CHECK IF ABOVE THRESHOLD
|
||||
# Optimization: collect threshold keys first to avoid sorting all model_info keys.
|
||||
# Exclude service_tier-specific variants (e.g. input_cost_per_token_above_200k_tokens_priority)
|
||||
# so that the threshold detection loop only processes standard keys. The
|
||||
# service_tier-specific above-threshold key is resolved later via _get_service_tier_cost_key.
|
||||
# Standard thresholds and thresholds suffixed for this request's service tier both count.
|
||||
tier_key_suffix: Final = _get_service_tier_cost_key("", service_tier)
|
||||
threshold_keys: Final = [
|
||||
k
|
||||
for k in model_info
|
||||
if k.startswith("input_cost_per_token_above_") and not k.endswith(_NON_STANDARD_THRESHOLD_SUFFIXES)
|
||||
if k.startswith("input_cost_per_token_above_")
|
||||
and (
|
||||
not k.endswith(_NON_STANDARD_THRESHOLD_SUFFIXES) or (tier_key_suffix != "" and k.endswith(tier_key_suffix))
|
||||
)
|
||||
]
|
||||
|
||||
# Only sort the threshold keys (typically 1-2 keys instead of 66+)
|
||||
|
|
|
|||
|
|
@ -87,6 +87,7 @@ from litellm.litellm_core_utils.get_llm_provider_logic import (
|
|||
is_registered_custom_provider,
|
||||
)
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLogging
|
||||
from litellm.litellm_core_utils.llm_cost_calc.utils import SERVICE_TIER_COST_KEY_SUFFIXES
|
||||
from litellm.litellm_core_utils.ptu_pricing import (
|
||||
PTU_COST_ATTRIBUTION_ENV_VAR,
|
||||
declares_ptu,
|
||||
|
|
@ -8812,6 +8813,41 @@ class Router:
|
|||
if backend_value is not None:
|
||||
model_info[field] = backend_value
|
||||
|
||||
@staticmethod
|
||||
def _cost_map_backend_model(deployment: Deployment) -> str:
|
||||
model_info_base_model: Final = deployment.model_info.base_model
|
||||
if isinstance(model_info_base_model, str) and model_info_base_model:
|
||||
return model_info_base_model
|
||||
params_base_model: Final = deployment.litellm_params.get("base_model")
|
||||
if isinstance(params_base_model, str) and params_base_model:
|
||||
return params_base_model
|
||||
return deployment.litellm_params.model
|
||||
|
||||
@staticmethod
|
||||
def _inherit_builtin_service_tier_pricing(
|
||||
model_info: dict, # mutable-ok: deployment cost-map entry filled in place
|
||||
backend_model: str,
|
||||
custom_llm_provider: str | None,
|
||||
) -> None:
|
||||
"""Inherit missing tier rates so a standalone entry does not fall back to custom standard rates."""
|
||||
if ptu_terms(model_info) is not None and is_ptu_cost_attribution_enabled():
|
||||
return
|
||||
if all(model_info.get(field) is None for field in ("input_cost_per_token", "output_cost_per_token")):
|
||||
return
|
||||
try:
|
||||
backend_info: Final = litellm.get_model_info(model=backend_model, custom_llm_provider=custom_llm_provider)
|
||||
except Exception: # noqa: BLE001 # get_model_info raises plain Exception for an unmapped backend model
|
||||
return
|
||||
backend_entry: Final = litellm.model_cost.get(backend_info.get("key") or "")
|
||||
if not isinstance(backend_entry, dict):
|
||||
return
|
||||
for field, backend_value in backend_entry.items():
|
||||
if not field.endswith(SERVICE_TIER_COST_KEY_SUFFIXES):
|
||||
continue
|
||||
if model_info.get(field) is not None or backend_value is None:
|
||||
continue
|
||||
model_info[field] = copy.deepcopy(backend_value)
|
||||
|
||||
@staticmethod
|
||||
def _inherit_builtin_base_rates_for_off_peak(
|
||||
model_info: dict, # mutable-ok: cost-map entry filled in place
|
||||
|
|
@ -8960,6 +8996,11 @@ class Router:
|
|||
backend_model=deployment.litellm_params.model,
|
||||
custom_llm_provider=deployment.litellm_params.custom_llm_provider,
|
||||
)
|
||||
Router._inherit_builtin_service_tier_pricing(
|
||||
model_info=_model_info,
|
||||
backend_model=Router._cost_map_backend_model(deployment),
|
||||
custom_llm_provider=deployment.litellm_params.custom_llm_provider,
|
||||
)
|
||||
Router._inherit_builtin_tiered_output_rate(
|
||||
model_info=_model_info,
|
||||
backend_model=deployment.litellm_params.model,
|
||||
|
|
@ -10000,6 +10041,11 @@ class Router:
|
|||
backend_model=deployment.litellm_params.model,
|
||||
custom_llm_provider=deployment.litellm_params.custom_llm_provider,
|
||||
)
|
||||
Router._inherit_builtin_service_tier_pricing(
|
||||
model_info=model_info,
|
||||
backend_model=Router._cost_map_backend_model(deployment),
|
||||
custom_llm_provider=deployment.litellm_params.custom_llm_provider,
|
||||
)
|
||||
Router._inherit_builtin_tiered_output_rate(
|
||||
model_info=model_info,
|
||||
backend_model=deployment.litellm_params.model,
|
||||
|
|
|
|||
|
|
@ -1,5 +1,6 @@
|
|||
import json
|
||||
import uuid
|
||||
from pathlib import Path
|
||||
from typing import Final, Literal
|
||||
|
||||
import httpx
|
||||
|
|
@ -260,3 +261,142 @@ def test_ultrafast_long_context_prompt_bills_ultrafast_long_context_rates(
|
|||
body: Final = object_value(observation["body"])
|
||||
assert body.get("service_tier") == service_tier, body
|
||||
assert not set(LONG_CONTEXT_PRICING).intersection(body), body
|
||||
|
||||
|
||||
BUNDLED_COST_MAP: Final = (
|
||||
Path(__file__).resolve().parents[3] / "litellm" / "model_prices_and_context_window_backup.json"
|
||||
)
|
||||
CUSTOM_STANDARD_INPUT_RATE: Final = 0.001
|
||||
CUSTOM_STANDARD_OUTPUT_RATE: Final = 0.002
|
||||
|
||||
|
||||
def _bundled_rate(model: str, field: str) -> float:
|
||||
rate: Final = object_value(JSON_OBJECT.validate_json(BUNDLED_COST_MAP.read_bytes())[model])[field]
|
||||
assert isinstance(rate, float) and rate > 0, f"{model}.{field} in {BUNDLED_COST_MAP.name}: {rate}"
|
||||
return rate
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("service_tier", "input_field", "output_field"),
|
||||
(
|
||||
("ultrafast", "input_cost_per_token_ultrafast", "output_cost_per_token_ultrafast"),
|
||||
(None, None, None),
|
||||
),
|
||||
ids=("ultrafast", "standard"),
|
||||
)
|
||||
def test_custom_standard_rates_bill_served_ultrafast_tier_at_the_catalog_tier_rate(
|
||||
gateway: Gateway, service_tier: str | None, input_field: str | None, output_field: str | None
|
||||
) -> None:
|
||||
input_rate: Final = CUSTOM_STANDARD_INPUT_RATE if input_field is None else _bundled_rate("gpt-6-astra", input_field)
|
||||
output_rate: Final = (
|
||||
CUSTOM_STANDARD_OUTPUT_RATE if output_field is None else _bundled_rate("gpt-6-astra", output_field)
|
||||
)
|
||||
with gateway.scenario() as scenario:
|
||||
scenario_id: Final = f"custom-standard-ultrafast-{uuid.uuid4().hex}"
|
||||
handle: Final = register_scenario(
|
||||
scenario_id,
|
||||
JsonResponse(
|
||||
content_type="application/json",
|
||||
body={
|
||||
"id": "chatcmpl-$UNIQUE_ID",
|
||||
"object": "chat.completion",
|
||||
"created": 1,
|
||||
"model": "gpt-6-astra",
|
||||
"choices": [
|
||||
{"index": 0, "message": {"role": "assistant", "content": "OK"}, "finish_reason": "stop"}
|
||||
],
|
||||
"usage": {"prompt_tokens": 1000, "completion_tokens": 100, "total_tokens": 1100},
|
||||
**({} if service_tier is None else {"service_tier": service_tier}),
|
||||
},
|
||||
),
|
||||
)
|
||||
scenario.cleanups.callback(delete_scenario, handle)
|
||||
model: Final = scenario.model(
|
||||
model="openai/gpt-6-astra",
|
||||
api_key=scenario_id,
|
||||
api_base=handle.api_base(),
|
||||
input_cost_per_token=CUSTOM_STANDARD_INPUT_RATE,
|
||||
output_cost_per_token=CUSTOM_STANDARD_OUTPUT_RATE,
|
||||
)
|
||||
response: Final = gateway.request(
|
||||
"POST",
|
||||
"/v1/chat/completions",
|
||||
{
|
||||
"model": model,
|
||||
"messages": [{"role": "user", "content": "OK"}],
|
||||
**({} if service_tier is None else {"service_tier": service_tier}),
|
||||
},
|
||||
key=scenario.key(),
|
||||
)
|
||||
assert response.status_code == 200, response.text
|
||||
expected: Final = 1000 * input_rate + 100 * output_rate
|
||||
assert float(response.headers["x-litellm-response-cost"]) == pytest.approx(expected, rel=1e-6), response.text
|
||||
request_id: Final = string_value(object_value(response.json())["id"])
|
||||
rows: Final = eventually(
|
||||
lambda: read_rows('SELECT spend FROM "LiteLLM_SpendLogs" WHERE request_id = %s', (request_id,)),
|
||||
lambda values: len(values) == 1,
|
||||
seconds=70,
|
||||
)
|
||||
assert float(rows[0]["spend"]) == pytest.approx(expected, rel=1e-6), rows
|
||||
|
||||
|
||||
def test_custom_standard_rates_bill_catalog_ultrafast_long_context_rates(gateway: Gateway) -> None:
|
||||
input_rate: Final = _bundled_rate("gpt-6-astra", "input_cost_per_token_above_272k_tokens_ultrafast")
|
||||
output_rate: Final = _bundled_rate("gpt-6-astra", "output_cost_per_token_above_272k_tokens_ultrafast")
|
||||
with gateway.scenario() as scenario:
|
||||
scenario_id: Final = f"custom-standard-ultrafast-long-context-{uuid.uuid4().hex}"
|
||||
handle: Final = register_scenario(
|
||||
scenario_id,
|
||||
JsonResponse(
|
||||
content_type="application/json",
|
||||
body={
|
||||
"id": "chatcmpl-$UNIQUE_ID",
|
||||
"object": "chat.completion",
|
||||
"created": 1,
|
||||
"model": "gpt-6-astra",
|
||||
"choices": [
|
||||
{"index": 0, "message": {"role": "assistant", "content": "OK"}, "finish_reason": "stop"}
|
||||
],
|
||||
"usage": {
|
||||
"prompt_tokens": LONG_PROMPT_TOKENS,
|
||||
"completion_tokens": 100,
|
||||
"total_tokens": LONG_PROMPT_TOKENS + 100,
|
||||
},
|
||||
"service_tier": "ultrafast",
|
||||
},
|
||||
),
|
||||
)
|
||||
scenario.cleanups.callback(delete_scenario, handle)
|
||||
model: Final = scenario.model(
|
||||
model="openai/gpt-6-astra",
|
||||
api_key=scenario_id,
|
||||
api_base=handle.api_base(),
|
||||
input_cost_per_token=CUSTOM_STANDARD_INPUT_RATE,
|
||||
output_cost_per_token=CUSTOM_STANDARD_OUTPUT_RATE,
|
||||
)
|
||||
response: Final = gateway.request(
|
||||
"POST",
|
||||
"/v1/chat/completions",
|
||||
{
|
||||
"model": model,
|
||||
"messages": [{"role": "user", "content": "long context ultrafast pricing"}],
|
||||
"service_tier": "ultrafast",
|
||||
},
|
||||
key=scenario.key(),
|
||||
)
|
||||
|
||||
assert response.status_code == 200, response.text
|
||||
expected: Final = LONG_PROMPT_TOKENS * input_rate + 100 * output_rate
|
||||
assert float(response.headers["x-litellm-response-cost"]) == pytest.approx(expected, rel=1e-6), response.text
|
||||
request_id: Final = string_value(object_value(response.json())["id"])
|
||||
rows: Final = eventually(
|
||||
lambda: read_rows(
|
||||
'SELECT spend, prompt_tokens, completion_tokens FROM "LiteLLM_SpendLogs" WHERE request_id = %s',
|
||||
(request_id,),
|
||||
),
|
||||
lambda values: len(values) == 1,
|
||||
seconds=70,
|
||||
)
|
||||
assert rows[0]["prompt_tokens"] == LONG_PROMPT_TOKENS
|
||||
assert rows[0]["completion_tokens"] == 100
|
||||
assert float(rows[0]["spend"]) == pytest.approx(expected, rel=1e-6), rows
|
||||
|
|
|
|||
|
|
@ -78,6 +78,64 @@ def test_completion_cost_bills_the_price_columns_of_the_service_tier(
|
|||
assert cost == pytest.approx(_cost_at(TIER_ROW, column_suffix))
|
||||
|
||||
|
||||
LONG_CONTEXT_TIER_MODEL: Final = "long-context-tier-priced-test-model"
|
||||
LONG_CONTEXT_TIER_ROW: Final[Mapping[str, float]] = MappingProxyType(
|
||||
{
|
||||
"input_cost_per_token": 4e-06,
|
||||
"output_cost_per_token": 8e-06,
|
||||
"input_cost_per_token_ultrafast": 1e-05,
|
||||
"output_cost_per_token_ultrafast": 2e-05,
|
||||
"input_cost_per_token_above_272k_tokens_ultrafast": 5e-05,
|
||||
"output_cost_per_token_above_272k_tokens_ultrafast": 6e-05,
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("service_tier", "prompt_tokens", "input_rate", "output_rate"),
|
||||
(
|
||||
pytest.param("ultrafast", 300_000, 5e-05, 6e-05, id="long-ultrafast"),
|
||||
pytest.param(None, 300_000, 4e-06, 8e-06, id="long-standard"),
|
||||
pytest.param("ultrafast", 1_000, 1e-05, 2e-05, id="short-ultrafast"),
|
||||
pytest.param("priority", 300_000, 4e-06, 8e-06, id="long-priority-falls-back"),
|
||||
),
|
||||
)
|
||||
def test_completion_cost_uses_only_the_request_tiers_long_context_rates(
|
||||
local_model_cost_map: None,
|
||||
service_tier: str | None,
|
||||
prompt_tokens: int,
|
||||
input_rate: float,
|
||||
output_rate: float,
|
||||
) -> None:
|
||||
litellm.register_model(
|
||||
{
|
||||
LONG_CONTEXT_TIER_MODEL: {
|
||||
"litellm_provider": "openai",
|
||||
"mode": "chat",
|
||||
**dict(LONG_CONTEXT_TIER_ROW),
|
||||
}
|
||||
}
|
||||
)
|
||||
completion_tokens: Final = 100
|
||||
response: Final = ModelResponse(
|
||||
model=LONG_CONTEXT_TIER_MODEL,
|
||||
usage=Usage(
|
||||
prompt_tokens=prompt_tokens,
|
||||
completion_tokens=completion_tokens,
|
||||
total_tokens=prompt_tokens + completion_tokens,
|
||||
),
|
||||
)
|
||||
|
||||
cost: Final = litellm.completion_cost(
|
||||
completion_response=response,
|
||||
model=LONG_CONTEXT_TIER_MODEL,
|
||||
custom_llm_provider="openai",
|
||||
service_tier=service_tier,
|
||||
)
|
||||
|
||||
assert cost == pytest.approx(prompt_tokens * input_rate + completion_tokens * output_rate)
|
||||
|
||||
|
||||
class _CostRecorder(CustomLogger):
|
||||
def __init__(self) -> None:
|
||||
super().__init__()
|
||||
|
|
|
|||
|
|
@ -23,6 +23,7 @@ from litellm import Router
|
|||
from litellm.caching.in_memory_cache import InMemoryCache
|
||||
from litellm.constants import DEFAULT_MAX_LRU_CACHE_SIZE
|
||||
from litellm.litellm_core_utils.ptu_pricing import ptu_config_error
|
||||
from litellm.litellm_core_utils.llm_cost_calc.utils import SERVICE_TIER_COST_KEY_SUFFIXES
|
||||
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler
|
||||
from litellm.llms.openai_like.model_info import MODEL_INFO_REFRESH_SECONDS
|
||||
from litellm.types.router import Deployment, LiteLLM_Params, ModelInfo
|
||||
|
|
@ -862,6 +863,425 @@ def test_inherit_builtin_cache_pricing_noop_for_unknown_backend():
|
|||
assert model_info == {"input_cost_per_token": 0.000003}
|
||||
|
||||
|
||||
_TIER_BACKEND_MODEL: Final = "tier-priced-backend"
|
||||
_TIER_BACKEND_KEY: Final = f"openai/{_TIER_BACKEND_MODEL}"
|
||||
_CUSTOM_STANDARD_INPUT_RATE: Final = 0.00011
|
||||
_CUSTOM_STANDARD_OUTPUT_RATE: Final = 0.00022
|
||||
_TIER_BACKEND_ENTRY: Final = {
|
||||
"key": _TIER_BACKEND_KEY,
|
||||
"litellm_provider": "openai",
|
||||
"mode": "chat",
|
||||
"max_tokens": 123456,
|
||||
"input_cost_per_token": 0.00021,
|
||||
"output_cost_per_token": 0.00032,
|
||||
"input_cost_per_token_ultrafast": 0.00031,
|
||||
"output_cost_per_token_ultrafast": 0.00042,
|
||||
"input_cost_per_token_priority": 0.00051,
|
||||
"output_cost_per_token_priority": 0.00062,
|
||||
"input_cost_per_token_flex": 0.00071,
|
||||
"output_cost_per_token_flex": 0.00082,
|
||||
"input_cost_per_token_balanced": 0.00091,
|
||||
"output_cost_per_token_balanced": 0.00102,
|
||||
"cache_read_input_token_cost_ultrafast": 0.00013,
|
||||
"input_cost_per_token_above_272k_tokens_ultrafast": 0.00014,
|
||||
"output_cost_per_token_above_272k_tokens_ultrafast": 0.00015,
|
||||
"input_cost_per_token_batches": 0.00016,
|
||||
"input_cost_per_token_above_272k_tokens": 0.00017,
|
||||
}
|
||||
_AZURE_TIER_BACKEND_KEY: Final = "azure/tier-priced-backend"
|
||||
_AZURE_TIER_BACKEND_ENTRY: Final = {
|
||||
**_TIER_BACKEND_ENTRY,
|
||||
"key": _AZURE_TIER_BACKEND_KEY,
|
||||
"litellm_provider": "azure",
|
||||
}
|
||||
|
||||
|
||||
def _register_tier_backend() -> None:
|
||||
litellm.model_cost[_TIER_BACKEND_KEY] = copy.deepcopy(_TIER_BACKEND_ENTRY)
|
||||
litellm.get_model_info.cache_clear()
|
||||
_invalidate_model_cost_lowercase_map()
|
||||
|
||||
|
||||
def _register_azure_tier_backend() -> None:
|
||||
litellm.model_cost[_AZURE_TIER_BACKEND_KEY] = copy.deepcopy(_AZURE_TIER_BACKEND_ENTRY)
|
||||
litellm.get_model_info.cache_clear()
|
||||
_invalidate_model_cost_lowercase_map()
|
||||
|
||||
|
||||
def test_inherit_builtin_service_tier_pricing_fills_only_missing_fields() -> None:
|
||||
model_cost_entries: Final = {
|
||||
key: copy.deepcopy(litellm.model_cost.get(key))
|
||||
for key in (_TIER_BACKEND_KEY, _TIER_BACKEND_MODEL)
|
||||
}
|
||||
try:
|
||||
_register_tier_backend()
|
||||
model_info: Final = {
|
||||
"id": "custom-priced-tier-deployment",
|
||||
"input_cost_per_token": _CUSTOM_STANDARD_INPUT_RATE,
|
||||
"output_cost_per_token": _CUSTOM_STANDARD_OUTPUT_RATE,
|
||||
"output_cost_per_token_ultrafast": 0.00999,
|
||||
}
|
||||
|
||||
Router._inherit_builtin_service_tier_pricing(
|
||||
model_info=model_info,
|
||||
backend_model=_TIER_BACKEND_MODEL,
|
||||
custom_llm_provider="openai",
|
||||
)
|
||||
|
||||
assert model_info == {
|
||||
"id": "custom-priced-tier-deployment",
|
||||
"input_cost_per_token": _CUSTOM_STANDARD_INPUT_RATE,
|
||||
"output_cost_per_token": _CUSTOM_STANDARD_OUTPUT_RATE,
|
||||
"input_cost_per_token_ultrafast": _TIER_BACKEND_ENTRY["input_cost_per_token_ultrafast"],
|
||||
"output_cost_per_token_ultrafast": 0.00999,
|
||||
"input_cost_per_token_priority": _TIER_BACKEND_ENTRY["input_cost_per_token_priority"],
|
||||
"output_cost_per_token_priority": _TIER_BACKEND_ENTRY["output_cost_per_token_priority"],
|
||||
"input_cost_per_token_flex": _TIER_BACKEND_ENTRY["input_cost_per_token_flex"],
|
||||
"output_cost_per_token_flex": _TIER_BACKEND_ENTRY["output_cost_per_token_flex"],
|
||||
"input_cost_per_token_balanced": _TIER_BACKEND_ENTRY["input_cost_per_token_balanced"],
|
||||
"output_cost_per_token_balanced": _TIER_BACKEND_ENTRY["output_cost_per_token_balanced"],
|
||||
"cache_read_input_token_cost_ultrafast": _TIER_BACKEND_ENTRY[
|
||||
"cache_read_input_token_cost_ultrafast"
|
||||
],
|
||||
"input_cost_per_token_above_272k_tokens_ultrafast": _TIER_BACKEND_ENTRY[
|
||||
"input_cost_per_token_above_272k_tokens_ultrafast"
|
||||
],
|
||||
"output_cost_per_token_above_272k_tokens_ultrafast": _TIER_BACKEND_ENTRY[
|
||||
"output_cost_per_token_above_272k_tokens_ultrafast"
|
||||
],
|
||||
}
|
||||
finally:
|
||||
_restore_model_cost_entries(model_cost_entries)
|
||||
litellm.get_model_info.cache_clear()
|
||||
|
||||
|
||||
def test_inherit_builtin_service_tier_pricing_noop_without_base_rate_or_backend() -> None:
|
||||
model_cost_entries: Final = {
|
||||
key: copy.deepcopy(litellm.model_cost.get(key))
|
||||
for key in (_TIER_BACKEND_KEY, _TIER_BACKEND_MODEL)
|
||||
}
|
||||
try:
|
||||
_register_tier_backend()
|
||||
model_info_without_base_rate: Final = {
|
||||
"id": "custom-priced-no-base-rate",
|
||||
"input_cost_per_token_ultrafast": 0.00031,
|
||||
}
|
||||
expected_without_base_rate: Final = copy.deepcopy(model_info_without_base_rate)
|
||||
Router._inherit_builtin_service_tier_pricing(
|
||||
model_info=model_info_without_base_rate,
|
||||
backend_model=_TIER_BACKEND_MODEL,
|
||||
custom_llm_provider="openai",
|
||||
)
|
||||
|
||||
model_info_with_unknown_backend: Final = {
|
||||
"id": "custom-priced-unknown-backend",
|
||||
"input_cost_per_token": _CUSTOM_STANDARD_INPUT_RATE,
|
||||
"output_cost_per_token": _CUSTOM_STANDARD_OUTPUT_RATE,
|
||||
}
|
||||
expected_with_unknown_backend: Final = copy.deepcopy(model_info_with_unknown_backend)
|
||||
Router._inherit_builtin_service_tier_pricing(
|
||||
model_info=model_info_with_unknown_backend,
|
||||
backend_model="tier-priced-backend-unknown",
|
||||
custom_llm_provider="openai",
|
||||
)
|
||||
|
||||
assert model_info_without_base_rate == expected_without_base_rate
|
||||
assert model_info_with_unknown_backend == expected_with_unknown_backend
|
||||
finally:
|
||||
_restore_model_cost_entries(model_cost_entries)
|
||||
litellm.get_model_info.cache_clear()
|
||||
|
||||
|
||||
def test_router_completion_uses_custom_standard_and_backend_ultrafast_pricing() -> None:
|
||||
model_id: Final = "tier-priced-deployment"
|
||||
model_cost_entries: Final = {
|
||||
key: copy.deepcopy(litellm.model_cost.get(key))
|
||||
for key in (_TIER_BACKEND_KEY, _TIER_BACKEND_MODEL, model_id)
|
||||
}
|
||||
try:
|
||||
_register_tier_backend()
|
||||
router: Final = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "tier-priced-router",
|
||||
"litellm_params": {
|
||||
"model": _TIER_BACKEND_MODEL,
|
||||
"custom_llm_provider": "openai",
|
||||
"api_key": "sk-tier-pricing-not-used",
|
||||
"input_cost_per_token": _CUSTOM_STANDARD_INPUT_RATE,
|
||||
"output_cost_per_token": _CUSTOM_STANDARD_OUTPUT_RATE,
|
||||
},
|
||||
"model_info": {
|
||||
"id": model_id,
|
||||
"input_cost_per_token": _CUSTOM_STANDARD_INPUT_RATE,
|
||||
"output_cost_per_token": _CUSTOM_STANDARD_OUTPUT_RATE,
|
||||
},
|
||||
}
|
||||
]
|
||||
)
|
||||
|
||||
ultrafast_response: Final = router.completion(
|
||||
model="tier-priced-router",
|
||||
messages=[{"role": "user", "content": "tiered pricing"}],
|
||||
service_tier="ultrafast",
|
||||
mock_response=litellm.ModelResponse(
|
||||
model=_TIER_BACKEND_MODEL,
|
||||
service_tier="ultrafast",
|
||||
usage=litellm.Usage(prompt_tokens=1000, completion_tokens=100, total_tokens=1100),
|
||||
),
|
||||
)
|
||||
standard_response: Final = router.completion(
|
||||
model="tier-priced-router",
|
||||
messages=[{"role": "user", "content": "standard pricing"}],
|
||||
mock_response=litellm.ModelResponse(
|
||||
model=_TIER_BACKEND_MODEL,
|
||||
usage=litellm.Usage(prompt_tokens=1000, completion_tokens=100, total_tokens=1100),
|
||||
),
|
||||
)
|
||||
|
||||
assert isinstance(ultrafast_response, litellm.ModelResponse)
|
||||
assert ultrafast_response._hidden_params["response_cost"] == pytest.approx(
|
||||
1000 * _TIER_BACKEND_ENTRY["input_cost_per_token_ultrafast"]
|
||||
+ 100 * _TIER_BACKEND_ENTRY["output_cost_per_token_ultrafast"]
|
||||
)
|
||||
assert isinstance(standard_response, litellm.ModelResponse)
|
||||
assert standard_response._hidden_params["response_cost"] == pytest.approx(
|
||||
1000 * _CUSTOM_STANDARD_INPUT_RATE + 100 * _CUSTOM_STANDARD_OUTPUT_RATE
|
||||
)
|
||||
finally:
|
||||
_restore_model_cost_entries(model_cost_entries)
|
||||
litellm.get_model_info.cache_clear()
|
||||
|
||||
|
||||
def test_router_completion_uses_backend_ultrafast_long_context_rates() -> None:
|
||||
model_id: Final = "tier-priced-long-context-deployment"
|
||||
model_cost_entries: Final = {
|
||||
key: copy.deepcopy(litellm.model_cost.get(key))
|
||||
for key in (_TIER_BACKEND_KEY, _TIER_BACKEND_MODEL, model_id)
|
||||
}
|
||||
try:
|
||||
_register_tier_backend()
|
||||
router: Final = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "tier-priced-long-context-router",
|
||||
"litellm_params": {
|
||||
"model": _TIER_BACKEND_MODEL,
|
||||
"custom_llm_provider": "openai",
|
||||
"api_key": "sk-tier-pricing-not-used",
|
||||
"input_cost_per_token": _CUSTOM_STANDARD_INPUT_RATE,
|
||||
"output_cost_per_token": _CUSTOM_STANDARD_OUTPUT_RATE,
|
||||
},
|
||||
"model_info": {
|
||||
"id": model_id,
|
||||
"input_cost_per_token": _CUSTOM_STANDARD_INPUT_RATE,
|
||||
"output_cost_per_token": _CUSTOM_STANDARD_OUTPUT_RATE,
|
||||
},
|
||||
}
|
||||
]
|
||||
)
|
||||
|
||||
response: Final = router.completion(
|
||||
model="tier-priced-long-context-router",
|
||||
messages=[{"role": "user", "content": "long context tiered pricing"}],
|
||||
service_tier="ultrafast",
|
||||
mock_response=litellm.ModelResponse(
|
||||
model=_TIER_BACKEND_MODEL,
|
||||
service_tier="ultrafast",
|
||||
usage=litellm.Usage(prompt_tokens=300_000, completion_tokens=100, total_tokens=300_100),
|
||||
),
|
||||
)
|
||||
|
||||
assert isinstance(response, litellm.ModelResponse)
|
||||
assert response._hidden_params["response_cost"] == pytest.approx(
|
||||
300_000 * _TIER_BACKEND_ENTRY["input_cost_per_token_above_272k_tokens_ultrafast"]
|
||||
+ 100 * _TIER_BACKEND_ENTRY["output_cost_per_token_above_272k_tokens_ultrafast"]
|
||||
)
|
||||
finally:
|
||||
_restore_model_cost_entries(model_cost_entries)
|
||||
litellm.get_model_info.cache_clear()
|
||||
|
||||
|
||||
@pytest.mark.parametrize("ptu_enabled", (True, False))
|
||||
def test_ptu_service_tier_pricing_is_disabled_only_when_attribution_is_enabled(
|
||||
monkeypatch: pytest.MonkeyPatch, ptu_enabled: bool
|
||||
) -> None:
|
||||
model_id: Final = f"ptu-tier-deployment-{ptu_enabled}"
|
||||
model_cost_entries: Final = {
|
||||
key: copy.deepcopy(litellm.model_cost.get(key))
|
||||
for key in (_TIER_BACKEND_KEY, model_id)
|
||||
}
|
||||
try:
|
||||
_register_tier_backend()
|
||||
monkeypatch.setenv("LITELLM_ENABLE_PTU_COST_ATTRIBUTION", "True" if ptu_enabled else "")
|
||||
router: Final = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": f"ptu-tier-model-{ptu_enabled}",
|
||||
"litellm_params": {
|
||||
"model": _TIER_BACKEND_MODEL,
|
||||
"custom_llm_provider": "openai",
|
||||
"api_key": "sk-tier-pricing-not-used",
|
||||
"input_cost_per_token": _CUSTOM_STANDARD_INPUT_RATE,
|
||||
"output_cost_per_token": _CUSTOM_STANDARD_OUTPUT_RATE,
|
||||
},
|
||||
"model_info": {**_PTU_MODEL_INFO, "id": model_id},
|
||||
}
|
||||
]
|
||||
)
|
||||
registered: Final = litellm.model_cost[model_id]
|
||||
tier_fields: Final = tuple(
|
||||
field for field in _TIER_BACKEND_ENTRY if field.endswith(SERVICE_TIER_COST_KEY_SUFFIXES)
|
||||
)
|
||||
if ptu_enabled:
|
||||
assert all(field not in registered for field in tier_fields)
|
||||
else:
|
||||
assert all(field in registered for field in tier_fields)
|
||||
|
||||
response: Final = router.completion(
|
||||
model=f"ptu-tier-model-{ptu_enabled}",
|
||||
messages=[{"role": "user", "content": "ptu service tier pricing"}],
|
||||
service_tier="priority",
|
||||
mock_response=litellm.ModelResponse(
|
||||
model=_TIER_BACKEND_MODEL,
|
||||
service_tier="priority",
|
||||
usage=litellm.Usage(prompt_tokens=1000, completion_tokens=100, total_tokens=1100),
|
||||
),
|
||||
)
|
||||
|
||||
assert isinstance(response, litellm.ModelResponse)
|
||||
expected_cost: Final = (
|
||||
0.0
|
||||
if ptu_enabled
|
||||
else 1000 * _TIER_BACKEND_ENTRY["input_cost_per_token_priority"]
|
||||
+ 100 * _TIER_BACKEND_ENTRY["output_cost_per_token_priority"]
|
||||
)
|
||||
assert response._hidden_params["response_cost"] == pytest.approx(expected_cost)
|
||||
finally:
|
||||
_restore_model_cost_entries(model_cost_entries)
|
||||
litellm.get_model_info.cache_clear()
|
||||
|
||||
|
||||
def test_azure_base_model_inherits_service_tier_pricing_for_registration_and_payload() -> None:
|
||||
model_id: Final = "azure-tier-priced-alias"
|
||||
payload_id: Final = "azure-tier-priced-payload"
|
||||
model_cost_entries: Final = {
|
||||
key: copy.deepcopy(litellm.model_cost.get(key))
|
||||
for key in (_AZURE_TIER_BACKEND_KEY, model_id, payload_id)
|
||||
}
|
||||
try:
|
||||
_register_azure_tier_backend()
|
||||
router: Final = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "azure/tier-priced-alias",
|
||||
"litellm_params": {
|
||||
"model": "azure/tier-priced-alias",
|
||||
"custom_llm_provider": "azure",
|
||||
"api_key": "sk-tier-pricing-not-used",
|
||||
"api_base": "https://tier-priced.azure.invalid",
|
||||
},
|
||||
"model_info": {
|
||||
"id": model_id,
|
||||
"base_model": _AZURE_TIER_BACKEND_KEY,
|
||||
"input_cost_per_token": _CUSTOM_STANDARD_INPUT_RATE,
|
||||
"output_cost_per_token": _CUSTOM_STANDARD_OUTPUT_RATE,
|
||||
},
|
||||
}
|
||||
]
|
||||
)
|
||||
|
||||
response: Final = router.completion(
|
||||
model="azure/tier-priced-alias",
|
||||
messages=[{"role": "user", "content": "azure base model pricing"}],
|
||||
service_tier="priority",
|
||||
allowed_openai_params=["service_tier"],
|
||||
mock_response=litellm.ModelResponse(
|
||||
model=_AZURE_TIER_BACKEND_KEY,
|
||||
service_tier="priority",
|
||||
usage=litellm.Usage(prompt_tokens=1000, completion_tokens=100, total_tokens=1100),
|
||||
),
|
||||
)
|
||||
|
||||
assert isinstance(response, litellm.ModelResponse)
|
||||
assert response._hidden_params["response_cost"] == pytest.approx(
|
||||
1000 * _AZURE_TIER_BACKEND_ENTRY["input_cost_per_token_priority"]
|
||||
+ 100 * _AZURE_TIER_BACKEND_ENTRY["output_cost_per_token_priority"]
|
||||
)
|
||||
|
||||
payload: Final = Router._deployment_model_cost_payload(
|
||||
deployment=Deployment(
|
||||
model_name="azure/tier-priced-alias-from-params",
|
||||
litellm_params=LiteLLM_Params(
|
||||
model="azure/tier-priced-alias",
|
||||
custom_llm_provider="azure",
|
||||
base_model=_AZURE_TIER_BACKEND_KEY,
|
||||
input_cost_per_token=_CUSTOM_STANDARD_INPUT_RATE,
|
||||
output_cost_per_token=_CUSTOM_STANDARD_OUTPUT_RATE,
|
||||
),
|
||||
model_info=ModelInfo(id=payload_id),
|
||||
)
|
||||
)
|
||||
|
||||
assert payload["input_cost_per_token_priority"] == _AZURE_TIER_BACKEND_ENTRY[
|
||||
"input_cost_per_token_priority"
|
||||
]
|
||||
assert payload["output_cost_per_token_priority"] == _AZURE_TIER_BACKEND_ENTRY[
|
||||
"output_cost_per_token_priority"
|
||||
]
|
||||
finally:
|
||||
_restore_model_cost_entries(model_cost_entries)
|
||||
litellm.get_model_info.cache_clear()
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("model_info_base_model", "params_base_model", "model", "expected"),
|
||||
(
|
||||
pytest.param(
|
||||
"azure/tier-priced-model-info-base",
|
||||
"azure/tier-priced-params-base",
|
||||
"azure/tier-priced-deployment-alias",
|
||||
"azure/tier-priced-model-info-base",
|
||||
id="model-info-base-model-wins",
|
||||
),
|
||||
pytest.param(
|
||||
None,
|
||||
"azure/tier-priced-params-base",
|
||||
"azure/tier-priced-deployment-alias",
|
||||
"azure/tier-priced-params-base",
|
||||
id="params-base-model-fallback",
|
||||
),
|
||||
pytest.param(
|
||||
None,
|
||||
None,
|
||||
"azure/tier-priced-deployment-alias",
|
||||
"azure/tier-priced-deployment-alias",
|
||||
id="model-fallback",
|
||||
),
|
||||
pytest.param(
|
||||
"",
|
||||
"azure/tier-priced-params-base",
|
||||
"azure/tier-priced-deployment-alias",
|
||||
"azure/tier-priced-params-base",
|
||||
id="empty-model-info-base-model-falls-through",
|
||||
),
|
||||
),
|
||||
)
|
||||
def test_cost_map_backend_model_uses_canonical_model_precedence(
|
||||
model_info_base_model: str | None,
|
||||
params_base_model: str | None,
|
||||
model: str,
|
||||
expected: str,
|
||||
) -> None:
|
||||
deployment: Final = Deployment(
|
||||
model_name="azure/tier-priced-cost-map-backend",
|
||||
litellm_params=LiteLLM_Params(model=model, base_model=params_base_model),
|
||||
model_info=ModelInfo(id="tier-priced-cost-map-backend", base_model=model_info_base_model),
|
||||
)
|
||||
|
||||
assert Router._cost_map_backend_model(deployment) == expected
|
||||
|
||||
|
||||
def test_inherit_builtin_base_rates_for_off_peak_fills_missing_rates():
|
||||
"""Direct unit test of the helper: an entry carrying only an
|
||||
off_peak_pricing block inherits the backend model's built-in base token
|
||||
|
|
@ -1803,6 +2223,41 @@ def test_deployment_model_cost_payload_folds_in_litellm_params_pricing():
|
|||
assert payload["cache_read_input_token_cost"] > 0
|
||||
|
||||
|
||||
def test_deployment_model_cost_payload_includes_builtin_service_tier_pricing() -> None:
|
||||
model_id: Final = "tier-priced-payload"
|
||||
model_cost_entries: Final = {
|
||||
key: copy.deepcopy(litellm.model_cost.get(key))
|
||||
for key in (_TIER_BACKEND_KEY, _TIER_BACKEND_MODEL, model_id)
|
||||
}
|
||||
try:
|
||||
_register_tier_backend()
|
||||
payload: Final = Router._deployment_model_cost_payload(
|
||||
deployment=Deployment(
|
||||
model_name="tier-priced-payload",
|
||||
litellm_params=LiteLLM_Params(
|
||||
model=_TIER_BACKEND_MODEL,
|
||||
custom_llm_provider="openai",
|
||||
input_cost_per_token=_CUSTOM_STANDARD_INPUT_RATE,
|
||||
output_cost_per_token=_CUSTOM_STANDARD_OUTPUT_RATE,
|
||||
),
|
||||
model_info=ModelInfo(id=model_id),
|
||||
)
|
||||
)
|
||||
|
||||
assert (
|
||||
payload["input_cost_per_token_ultrafast"] == _TIER_BACKEND_ENTRY["input_cost_per_token_ultrafast"]
|
||||
)
|
||||
assert (
|
||||
payload["output_cost_per_token_ultrafast"] == _TIER_BACKEND_ENTRY["output_cost_per_token_ultrafast"]
|
||||
)
|
||||
assert payload["input_cost_per_token_balanced"] == _TIER_BACKEND_ENTRY["input_cost_per_token_balanced"]
|
||||
assert payload["input_cost_per_token"] == _CUSTOM_STANDARD_INPUT_RATE
|
||||
assert payload["output_cost_per_token"] == _CUSTOM_STANDARD_OUTPUT_RATE
|
||||
finally:
|
||||
_restore_model_cost_entries(model_cost_entries)
|
||||
litellm.get_model_info.cache_clear()
|
||||
|
||||
|
||||
def test_register_deployment_in_model_cost_writes_both_key_families():
|
||||
"""
|
||||
A deployment contributes its full model_info under its unique id and the
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue