mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-11 03:38:38 +00:00
Merge 1955d41beb into e1d16f51d1
This commit is contained in:
commit
b89ca50aa4
2 changed files with 126 additions and 3 deletions
|
|
@ -803,11 +803,18 @@ def _get_hidden_str_for_cost_calc(hidden_params: object, key: str) -> str | None
|
|||
_NON_TOKEN_RATE_FIELDS: Final = frozenset(
|
||||
{"cost_per_second", "input_cost_per_second", "output_cost_per_second", "input_cost_per_query", "tiered_pricing"}
|
||||
)
|
||||
# Count only when picking a deployment's own entry: speech bills per character, realtime bills per token
|
||||
_PER_CHARACTER_RATE_FIELDS: Final = frozenset({"input_cost_per_character", "output_cost_per_character"})
|
||||
|
||||
|
||||
def _cost_map_entry_prices_anything(entry: Mapping[str, object]) -> bool:
|
||||
def _cost_map_entry_prices_anything(
|
||||
entry: Mapping[str, object], extra_rate_fields: frozenset[str] = frozenset()
|
||||
) -> bool:
|
||||
return any(
|
||||
value is not None and (field in _NON_TOKEN_RATE_FIELDS or ("cost_per" in field and "token" in field))
|
||||
value is not None
|
||||
and (
|
||||
field in _NON_TOKEN_RATE_FIELDS or field in extra_rate_fields or ("cost_per" in field and "token" in field)
|
||||
)
|
||||
for field, value in entry.items()
|
||||
)
|
||||
|
||||
|
|
@ -850,7 +857,7 @@ def _select_model_name_for_cost_calc(
|
|||
if custom_pricing is True:
|
||||
if router_model_id is not None and router_model_id in litellm.model_cost:
|
||||
entry: Final = litellm.model_cost[router_model_id]
|
||||
if _cost_map_entry_prices_anything(entry):
|
||||
if _cost_map_entry_prices_anything(entry, extra_rate_fields=_PER_CHARACTER_RATE_FIELDS):
|
||||
return_model = router_model_id
|
||||
else:
|
||||
return_model = model
|
||||
|
|
|
|||
|
|
@ -1068,6 +1068,122 @@ def test_per_query_priced_rerank_deployment_completion_cost_is_nonzero():
|
|||
assert cost == pytest.approx(3 * 0.001)
|
||||
|
||||
|
||||
def test_per_character_priced_speech_deployment_bills_its_own_rate(_local_model_cost_map: None) -> None:
|
||||
from litellm import Router
|
||||
|
||||
rate: Final = 1e-8
|
||||
prompt: Final = "abcdefghijklm"
|
||||
router: Final = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "custom-tts",
|
||||
"litellm_params": {
|
||||
"model": "openai/custom-tts-unmapped",
|
||||
"api_key": "sk-fake",
|
||||
"input_cost_per_character": rate,
|
||||
},
|
||||
},
|
||||
]
|
||||
)
|
||||
router_model_id: Final = router.model_list[0]["model_info"]["id"]
|
||||
|
||||
cost: Final = completion_cost(
|
||||
completion_response=None,
|
||||
model="openai/custom-tts-unmapped",
|
||||
custom_llm_provider="openai",
|
||||
call_type="speech",
|
||||
prompt=prompt,
|
||||
custom_pricing=True,
|
||||
router_model_id=router_model_id,
|
||||
)
|
||||
|
||||
assert cost == pytest.approx(len(prompt) * rate)
|
||||
|
||||
|
||||
def test_per_character_rate_on_mapped_speech_deployment_beats_the_public_rate(_local_model_cost_map: None) -> None:
|
||||
from litellm import Router
|
||||
|
||||
public_rate: Final = litellm.model_cost["tts-1"]["input_cost_per_character"]
|
||||
deployment_rate: Final = public_rate / 3
|
||||
prompt: Final = "abcdefghijklm"
|
||||
router: Final = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "discounted-tts",
|
||||
"litellm_params": {
|
||||
"model": "openai/tts-1",
|
||||
"api_key": "sk-fake",
|
||||
"input_cost_per_character": deployment_rate,
|
||||
},
|
||||
},
|
||||
]
|
||||
)
|
||||
router_model_id: Final = router.model_list[0]["model_info"]["id"]
|
||||
|
||||
cost: Final = completion_cost(
|
||||
completion_response=None,
|
||||
model="openai/tts-1",
|
||||
custom_llm_provider="openai",
|
||||
call_type="speech",
|
||||
prompt=prompt,
|
||||
custom_pricing=True,
|
||||
router_model_id=router_model_id,
|
||||
)
|
||||
|
||||
assert cost == pytest.approx(len(prompt) * deployment_rate)
|
||||
assert cost != pytest.approx(len(prompt) * public_rate)
|
||||
|
||||
|
||||
def test_per_character_rate_on_realtime_deployment_keeps_session_token_pricing(
|
||||
_local_model_cost_map: None, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
"""Realtime is billed per token, so a character rate alone must not stop the
|
||||
lookup at the deployment and bill the session nothing."""
|
||||
from litellm.types.utils import CompletionTokensDetailsWrapper
|
||||
|
||||
model: Final = "gpt-realtime"
|
||||
deployment_key: Final = "deployment-id-for-a-character-priced-realtime-group"
|
||||
monkeypatch.setitem(
|
||||
litellm.model_cost,
|
||||
deployment_key,
|
||||
{"litellm_provider": "openai", "mode": "realtime", "input_cost_per_character": 1e-6},
|
||||
)
|
||||
logging_object: Final = LiteLLMRealtimeStreamLoggingObject(
|
||||
usage=Usage(
|
||||
prompt_tokens=120,
|
||||
completion_tokens=60,
|
||||
total_tokens=180,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(text_tokens=20, audio_tokens=100),
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(text_tokens=10, audio_tokens=50),
|
||||
),
|
||||
results=[
|
||||
{"type": "session.created", "session": {"model": model}},
|
||||
{
|
||||
"type": "response.done",
|
||||
"response": {"usage": {"input_tokens": 120, "output_tokens": 60, "total_tokens": 180}},
|
||||
},
|
||||
],
|
||||
)
|
||||
|
||||
public_cost: Final = completion_cost(
|
||||
completion_response=logging_object,
|
||||
model=model,
|
||||
call_type=CallTypes.arealtime.value,
|
||||
custom_llm_provider="openai",
|
||||
)
|
||||
deployment_cost: Final = completion_cost(
|
||||
completion_response=logging_object,
|
||||
model=model,
|
||||
call_type=CallTypes.arealtime.value,
|
||||
custom_llm_provider="openai",
|
||||
custom_pricing=True,
|
||||
router_model_id=deployment_key,
|
||||
)
|
||||
|
||||
assert public_cost > 0
|
||||
assert deployment_cost == pytest.approx(public_cost, rel=1e-9)
|
||||
|
||||
|
||||
def test_azure_realtime_cost_calculator(_local_model_cost_map):
|
||||
|
||||
cost = handle_realtime_stream_cost_calculation(
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue