mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
fix(cost): recognize character rates in deployment cost-map selection
A deployment whose only rates are input_cost_per_character / output_cost_per_character (e.g. a custom OpenAI-compatible TTS backend) failed _cost_map_entry_prices_anything(): the character fields are neither in _NON_TOKEN_RATE_FIELDS nor token-named, so the "prices anything" check returned False. With custom_pricing=True the router_model_id was then never selected and audio_speech (aspeech) resolved its price through the bare model name — which is not in the cost map — producing spend = 0 and no x-litellm-response-cost header (issue #44200). Add both character fields to _NON_TOKEN_RATE_FIELDS so character-only deployments are selected like per-second and per-query deployments.
This commit is contained in:
parent
3cf5bb2753
commit
1f5005ec54
2 changed files with 50 additions and 18 deletions
|
|
@ -800,7 +800,15 @@ def _get_hidden_str_for_cost_calc(hidden_params: object, key: str) -> str | None
|
|||
|
||||
|
||||
_NON_TOKEN_RATE_FIELDS: Final = frozenset(
|
||||
{"cost_per_second", "input_cost_per_second", "output_cost_per_second", "input_cost_per_query", "tiered_pricing"}
|
||||
{
|
||||
"cost_per_second",
|
||||
"input_cost_per_second",
|
||||
"output_cost_per_second",
|
||||
"input_cost_per_query",
|
||||
"input_cost_per_character",
|
||||
"output_cost_per_character",
|
||||
"tiered_pricing",
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -186,7 +186,9 @@ def test_response_cost_calculator_keeps_optional_params_out_of_hidden_params():
|
|||
assert optional_params["aws_session_token"] == "session-secret"
|
||||
|
||||
|
||||
def test_embedding_success_logging_and_spend_log_carry_no_forwarded_credentials(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
def test_embedding_success_logging_and_spend_log_carry_no_forwarded_credentials(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
from litellm.proxy import proxy_server
|
||||
from litellm.proxy.spend_tracking.spend_tracking_utils import _get_proxy_server_request_for_spend_logs_payload
|
||||
|
||||
|
|
@ -235,10 +237,6 @@ def test_embedding_success_logging_and_spend_log_carry_no_forwarded_credentials(
|
|||
assert logging_obj.optional_params["extra_headers"] == {"x-goog-api-key": "goog-secret"}
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
def test_realtime_stream_combines_text_and_audio_token_details():
|
||||
"""Realtime response.done usage with input_token_details / output_token_details."""
|
||||
from litellm.cost_calculator import RealtimeAPITokenUsageProcessor
|
||||
|
|
@ -1353,8 +1351,6 @@ def test_bedrock_cost_calculator_comparison_with_without_cache():
|
|||
print(f"Cost with cache: {cost_with_cache}")
|
||||
|
||||
|
||||
|
||||
|
||||
def test_gemini_25_explicit_caching_cost_direct_usage():
|
||||
"""
|
||||
Test that Gemini 2.5 models correctly calculate costs with explicit caching.
|
||||
|
|
@ -1923,8 +1919,6 @@ def test_cost_margin_with_discount(monkeypatch):
|
|||
print(f" - Expected: ${expected_cost:.6f}")
|
||||
|
||||
|
||||
|
||||
|
||||
def test_completion_cost_extracts_service_tier_from_response(_local_model_cost_map):
|
||||
"""Test that completion_cost extracts service_tier from completion_response object."""
|
||||
from litellm import completion_cost
|
||||
|
|
@ -2674,8 +2668,6 @@ def test_gemini_without_cache_tokens_details():
|
|||
print("✅ Gemini without cacheTokensDetails works correctly")
|
||||
|
||||
|
||||
|
||||
|
||||
def test_additional_costs_only_for_azure_ai(_local_model_cost_map):
|
||||
"""
|
||||
Test that _get_additional_costs is only called for azure_ai provider.
|
||||
|
|
@ -3220,9 +3212,7 @@ def test_cost_per_token_resolves_per_second_rate_precedence(
|
|||
|
||||
model: Final = "test-chat-per-second-rate-precedence"
|
||||
entry: Final = {**pricing_fields, "litellm_provider": "together_ai", "mode": "chat"}
|
||||
litellm.register_model(
|
||||
model_cost={model: entry}
|
||||
)
|
||||
litellm.register_model(model_cost={model: entry})
|
||||
|
||||
assert cost_per_token(
|
||||
model=model,
|
||||
|
|
@ -3647,6 +3637,42 @@ def test_combine_usage_objects_sums_mirrored_cache_write_fields_once():
|
|||
assert combined_pair.prompt_tokens_details.cache_creation_tokens == 100
|
||||
|
||||
|
||||
def test_select_model_name_selects_character_priced_deployment(_local_model_cost_map):
|
||||
"""
|
||||
A deployment whose only rate is input_cost_per_character must be selected
|
||||
by router_model_id: the aspeech cost path resolves its price through the
|
||||
deployment entry, and a character-only entry failing the "prices anything"
|
||||
check silently produced spend = 0 (issue #44200).
|
||||
"""
|
||||
from litellm.cost_calculator import _select_model_name_for_cost_calc
|
||||
|
||||
router_model_id = "openai/qwen-audio-3.1-tts-flash-uuid"
|
||||
litellm.model_cost[router_model_id] = {
|
||||
"input_cost_per_character": 1e-8,
|
||||
"output_cost_per_character": 0.0,
|
||||
"litellm_provider": "openai",
|
||||
}
|
||||
|
||||
selected = _select_model_name_for_cost_calc(
|
||||
model="qwen-audio-3.1-tts-flash",
|
||||
completion_response=None,
|
||||
custom_pricing=True,
|
||||
custom_llm_provider="openai",
|
||||
router_model_id=router_model_id,
|
||||
)
|
||||
|
||||
assert selected == router_model_id
|
||||
|
||||
|
||||
def test_cost_map_entry_prices_anything_recognizes_character_rates():
|
||||
from litellm.cost_calculator import _cost_map_entry_prices_anything
|
||||
|
||||
assert _cost_map_entry_prices_anything({"input_cost_per_character": 1e-8}) is True
|
||||
assert _cost_map_entry_prices_anything({"output_cost_per_character": 0.0}) is True
|
||||
assert _cost_map_entry_prices_anything({"input_cost_per_token": 1e-6}) is True
|
||||
assert _cost_map_entry_prices_anything({"mode": "audio_speech"}) is False
|
||||
|
||||
|
||||
def test_select_model_name_strips_unregistered_alias_prefix(_local_model_cost_map):
|
||||
"""A router-facing model_name alias containing "/" whose leading segment is NOT a
|
||||
registered provider must not be double-prefixed into a non-existent cost key.
|
||||
|
|
@ -4785,9 +4811,7 @@ def test_xai_batch_tier_discounts_the_long_context_rate_like_the_flat_batch_rate
|
|||
assert info[f"{prefix}_above_200k_tokens_batches"] < info[f"{prefix}_above_200k_tokens"]
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("prompt_tokens", "tier"), [(200_000, "_above_200k_tokens_batches"), (199_999, "_batches")]
|
||||
)
|
||||
@pytest.mark.parametrize(("prompt_tokens", "tier"), [(200_000, "_above_200k_tokens_batches"), (199_999, "_batches")])
|
||||
def test_xai_batch_cost_calculator_bills_the_200k_batch_tier_inclusively(
|
||||
_local_model_cost_map: None, prompt_tokens: int, tier: str
|
||||
) -> None:
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue