fix(tests): merge base tests and our cache-read regression tests

The add/add conflict on test_lowest_cost.py arose because:
- base added framework tests for cost-key isolation and zero-division fix
- our branch added two regression tests for the cache-read price tie-break

Resolution: base tests first, our two regression tests appended. All 9 pass.
This commit is contained in:
LancyZhao 2026-09-14 11:18:43 +08:00
parent c4bb6ad2a0
commit 4fc8a644a8
No known key found for this signature in database

View file

@ -101,3 +101,84 @@ async def test_async_log_success_event_counts_a_response_with_no_completion_toke
)
assert _recorded_minute_counters(cache) == {"tpm": 12, "rpm": 1}
@pytest.mark.parametrize("cheaper_cache_first", [True, False])
@pytest.mark.asyncio
async def test_cost_routing_breaks_input_output_tie_on_cache_read_cost(cheaper_cache_first):
"""
Regression test for https://github.com/BerriAI/litellm/issues/38064
Two deployments with identical input+output price must be separated by their
cache-read price, not by whichever one happens to be listed first. The pricier
deployment omits a cache-read price to exercise the input-cost fallback.
"""
cheaper = {
"model_name": "cache-tie-test",
"litellm_params": {
"model": "openai/tie-model-not-in-cost-map",
"input_cost_per_token": 1e-06,
"output_cost_per_token": 2e-06,
"cache_read_input_token_cost": 1e-08,
},
"model_info": {"id": "cheaper-cache"},
}
pricier = {
"model_name": "cache-tie-test",
"litellm_params": {
"model": "openai/tie-model-not-in-cost-map",
"input_cost_per_token": 1e-06,
"output_cost_per_token": 2e-06,
},
"model_info": {"id": "pricier-cache"},
}
model_list = [cheaper, pricier] if cheaper_cache_first else [pricier, cheaper]
logger = LowestCostLoggingHandler(router_cache=DualCache())
selected = await logger.async_get_available_deployments(
model_group="cache-tie-test", healthy_deployments=model_list
)
assert selected["model_info"]["id"] == "cheaper-cache"
@pytest.mark.parametrize("free_cache_first", [True, False])
@pytest.mark.asyncio
async def test_cost_routing_honors_zero_deployment_cache_read_cost(free_cache_first):
"""
A deployment-level cache_read_input_token_cost of 0 is a price, not a missing value.
Truthiness checks treat it as unset and fall through to the cost map / input-cost
fallback, which ranks a deployment whose cache reads are free as the priciest one
in a tie. Providers that do not charge for cache reads make this a real config.
"""
free = {
"model_name": "cache-zero-test",
"litellm_params": {
"model": "openai/zero-model-not-in-cost-map",
"input_cost_per_token": 1e-06,
"output_cost_per_token": 2e-06,
"cache_read_input_token_cost": 0.0,
},
"model_info": {"id": "free-cache"},
}
paid = {
"model_name": "cache-zero-test",
"litellm_params": {
"model": "openai/zero-model-not-in-cost-map",
"input_cost_per_token": 1e-06,
"output_cost_per_token": 2e-06,
"cache_read_input_token_cost": 1e-08,
},
"model_info": {"id": "paid-cache"},
}
model_list = [free, paid] if free_cache_first else [paid, free]
logger = LowestCostLoggingHandler(router_cache=DualCache())
selected = await logger.async_get_available_deployments(
model_group="cache-zero-test", healthy_deployments=model_list
)
assert selected["model_info"]["id"] == "free-cache"