mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-05 02:41:56 +00:00
test(cost_tracking): expect cache reads without a map rate to estimate at the input rate
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
88d1eb3b35
commit
af61d9247c
1 changed files with 7 additions and 9 deletions
|
|
@ -975,9 +975,9 @@ class TestEstimateCostCacheAndReasoningTokens:
|
|||
|
||||
@pytest.mark.asyncio
|
||||
async def test_a_model_without_cache_or_reasoning_prices_estimates_what_the_proxy_bills(self, monkeypatch):
|
||||
"""The cost calculator bills cache reads of a cost-map model without cache prices at zero,
|
||||
its cache writes at the input rate, and its reasoning tokens at the output rate. The estimate
|
||||
reports those effective rates."""
|
||||
"""The cost calculator bills cache reads and writes of a cost-map model without cache prices
|
||||
at the input rate, and its reasoning tokens at the output rate. The estimate reports those
|
||||
effective rates."""
|
||||
monkeypatch.setitem(
|
||||
litellm.model_cost,
|
||||
A_MAPPED_MODEL,
|
||||
|
|
@ -986,14 +986,12 @@ class TestEstimateCostCacheAndReasoningTokens:
|
|||
|
||||
response = await _estimate_with_cache_and_reasoning(None, model=A_MAPPED_MODEL)
|
||||
|
||||
assert response.cache_read_cost_per_request == 0.0
|
||||
assert response.cache_read_cost_per_request == pytest.approx(CACHE_READ_TOKENS * 5e-6)
|
||||
assert response.cache_creation_cost_per_request == pytest.approx(CACHE_CREATION_TOKENS * 5e-6)
|
||||
assert response.reasoning_cost_per_request == pytest.approx(REASONING_TOKENS * 6e-6)
|
||||
assert response.input_cost_per_request == pytest.approx((TEXT_INPUT_TOKENS + CACHE_CREATION_TOKENS) * 5e-6)
|
||||
assert response.cost_per_request == pytest.approx(
|
||||
(TEXT_INPUT_TOKENS + CACHE_CREATION_TOKENS) * 5e-6 + OUTPUT_TOKENS * 6e-6
|
||||
)
|
||||
assert response.cache_read_input_token_cost == 0.0
|
||||
assert response.input_cost_per_request == pytest.approx(INPUT_TOKENS * 5e-6)
|
||||
assert response.cost_per_request == pytest.approx(INPUT_TOKENS * 5e-6 + OUTPUT_TOKENS * 6e-6)
|
||||
assert response.cache_read_input_token_cost == pytest.approx(5e-6)
|
||||
assert response.cache_creation_input_token_cost == pytest.approx(5e-6)
|
||||
assert response.output_cost_per_reasoning_token == pytest.approx(6e-6)
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue