mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
Merge fd60471a84 into f445e466b4
This commit is contained in:
commit
bd1fbc74e2
2 changed files with 105 additions and 1 deletions
|
|
@ -812,6 +812,33 @@ def _cost_map_entry_prices_anything(entry: Mapping[str, object]) -> bool:
|
|||
)
|
||||
|
||||
|
||||
def _has_rate(model: str, custom_llm_provider: str | None) -> bool:
|
||||
# A registered entry is a real answer even when it prices at zero; an unregistered
|
||||
# model is not, because `get_model_info` invents zero rates for it rather than
|
||||
# raising. Hence truthiness here, not `is not None`.
|
||||
if model in litellm.model_cost:
|
||||
return True
|
||||
try:
|
||||
info: Final = litellm.get_model_info(model=model, custom_llm_provider=custom_llm_provider)
|
||||
except Exception: # noqa: BLE001 # get_model_info raises bare Exception for some unmapped models
|
||||
return False
|
||||
return any(info.get(key) for key in ("input_cost_per_token", "output_cost_per_token", "input_cost_per_second"))
|
||||
|
||||
|
||||
def _priced_provider_response_model(
|
||||
provider_response_model: str | None,
|
||||
completion_response_model: str | None,
|
||||
custom_llm_provider: str | None,
|
||||
) -> str | None:
|
||||
if provider_response_model is None:
|
||||
return None
|
||||
if _has_rate(provider_response_model, custom_llm_provider):
|
||||
return provider_response_model
|
||||
if completion_response_model is not None and _has_rate(completion_response_model, custom_llm_provider):
|
||||
return completion_response_model
|
||||
return provider_response_model
|
||||
|
||||
|
||||
def _select_model_name_for_cost_calc(
|
||||
model: str | None,
|
||||
completion_response: object | None,
|
||||
|
|
@ -858,7 +885,15 @@ def _select_model_name_for_cost_calc(
|
|||
return_model = model
|
||||
|
||||
elif base_model is not None or provider_response_model is not None:
|
||||
return_model = base_model if base_model is not None else provider_response_model
|
||||
return_model = (
|
||||
base_model
|
||||
if base_model is not None
|
||||
else _priced_provider_response_model(
|
||||
provider_response_model,
|
||||
completion_response_model if isinstance(completion_response_model, str) else None,
|
||||
custom_llm_provider,
|
||||
)
|
||||
)
|
||||
|
||||
elif completion_response_model is None and hidden_params is not None:
|
||||
if hidden_params.get("model", None) is not None and len(hidden_params["model"]) > 0:
|
||||
|
|
|
|||
|
|
@ -5639,3 +5639,72 @@ def test_completion_cost_bills_base_when_gemini_serves_on_demand(
|
|||
)
|
||||
|
||||
assert cost == pytest.approx(100 * 0.001 + 50 * 0.002)
|
||||
|
||||
|
||||
def test_an_unpriced_provider_slug_does_not_bill_the_turn_at_zero():
|
||||
"""Regression test for #42161.
|
||||
|
||||
`_select_model_name_for_cost_calc` prefers `provider_response_model` over
|
||||
`response.model`, and `get_model_info` answers a model it has never heard of with
|
||||
zero rates instead of raising, so a turn reporting an unregistered slug was billed
|
||||
0.0 with nothing in the logs to explain it. Anthropic hits this by naming the dated
|
||||
build in `message_start` while the map carries the family.
|
||||
|
||||
The invariant: what the provider reports must not change the price of a turn the
|
||||
price map cannot quote it by.
|
||||
"""
|
||||
from litellm.cost_calculator import completion_cost
|
||||
from litellm.types.utils import Choices, Message, ModelResponse, Usage
|
||||
|
||||
unpriced = "claude-test-42161"
|
||||
assert unpriced not in litellm.model_cost
|
||||
|
||||
def _response(reported: str | None) -> ModelResponse:
|
||||
response = ModelResponse(
|
||||
model="anthropic/claude-opus-5",
|
||||
choices=[Choices(message=Message(content="x"))],
|
||||
)
|
||||
response.usage = Usage(prompt_tokens=38, completion_tokens=24, total_tokens=62)
|
||||
response._hidden_params = {} if reported is None else {"provider_response_model": reported}
|
||||
return response
|
||||
|
||||
baseline = completion_cost(completion_response=_response(None), custom_llm_provider="anthropic")
|
||||
assert baseline > 0
|
||||
|
||||
reported_unpriced = completion_cost(completion_response=_response(unpriced), custom_llm_provider="anthropic")
|
||||
assert reported_unpriced == baseline
|
||||
|
||||
|
||||
def test_a_priced_provider_slug_still_decides_the_rate():
|
||||
"""The fallback must not cost the reported name its precedence.
|
||||
|
||||
When the provider names something the map does price, that is the most specific
|
||||
truth about what served the turn and it still decides the rate.
|
||||
"""
|
||||
from litellm.cost_calculator import completion_cost
|
||||
from litellm.types.utils import Choices, Message, ModelResponse, Usage
|
||||
|
||||
priced = "test-model-priced-42161"
|
||||
litellm.register_model(
|
||||
model_cost={
|
||||
priced: {
|
||||
"input_cost_per_token": 1e-05,
|
||||
"output_cost_per_token": 2e-05,
|
||||
"litellm_provider": "anthropic",
|
||||
"mode": "chat",
|
||||
}
|
||||
}
|
||||
)
|
||||
|
||||
def _response(reported: str | None, model: str) -> ModelResponse:
|
||||
response = ModelResponse(model=model, choices=[Choices(message=Message(content="x"))])
|
||||
response.usage = Usage(prompt_tokens=38, completion_tokens=24, total_tokens=62)
|
||||
response._hidden_params = {} if reported is None else {"provider_response_model": reported}
|
||||
return response
|
||||
|
||||
reported = completion_cost(
|
||||
completion_response=_response(priced, "anthropic/claude-opus-5"),
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
expected = 38 * 1e-05 + 24 * 2e-05
|
||||
assert reported == pytest.approx(expected)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue