fix(cost): bill per-character rates set on a deployment

A deployment priced only through input_cost_per_character or output_cost_per_character was treated as unpriced, so cost lookup fell back to the shared model key. The router strips custom pricing from that key, so audio_speech calls on such a deployment logged spend 0 and sent no x-litellm-response-cost header

Count both character rates as pricing when picking the deployment's own cost entry

Fixes #44200
This commit is contained in:
Darsh Joshi 2026-10-02 19:01:49 -04:00
parent aef0a53837
commit 49cd17ee10
2 changed files with 41 additions and 1 deletions

View file

@ -800,7 +800,15 @@ def _get_hidden_str_for_cost_calc(hidden_params: object, key: str) -> str | None
_NON_TOKEN_RATE_FIELDS: Final = frozenset(
{"cost_per_second", "input_cost_per_second", "output_cost_per_second", "input_cost_per_query", "tiered_pricing"}
{
"cost_per_second",
"input_cost_per_second",
"output_cost_per_second",
"input_cost_per_query",
"input_cost_per_character",
"output_cost_per_character",
"tiered_pricing",
}
)

View file

@ -1067,6 +1067,38 @@ def test_per_query_priced_rerank_deployment_completion_cost_is_nonzero():
assert cost == pytest.approx(3 * 0.001)
def test_per_character_priced_speech_deployment_bills_its_own_rate(_local_model_cost_map: None) -> None:
from litellm import Router
rate: Final = 1e-8
prompt: Final = "abcdefghijklm"
router: Final = Router(
model_list=[
{
"model_name": "custom-tts",
"litellm_params": {
"model": "openai/custom-tts-unmapped",
"api_key": "sk-fake",
"input_cost_per_character": rate,
},
},
]
)
router_model_id: Final = router.model_list[0]["model_info"]["id"]
cost: Final = completion_cost(
completion_response=None,
model="openai/custom-tts-unmapped",
custom_llm_provider="openai",
call_type="speech",
prompt=prompt,
custom_pricing=True,
router_model_id=router_model_id,
)
assert cost == pytest.approx(len(prompt) * rate)
def test_azure_realtime_cost_calculator(_local_model_cost_map):
cost = handle_realtime_stream_cost_calculation(