Merge pull request #38422 from BerriAI/litellm_gemini35_flashlite_flex_cache_price

fix(model_prices): correct gemini-3.5-flash-lite flex cache-read pricing
This commit is contained in:
Mateo Wang 2026-08-26 17:11:07 -07:00 • committed by GitHub
commit 4e295e8eb9
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
3 changed files with 51 additions and 4 deletions

View file

@ -20150,7 +20150,7 @@
"gemini-3.5-flash-lite": {
"deprecation_date": "2027-07-21",
"cache_read_input_token_cost": 3e-08,
"cache_read_input_token_cost_flex": 2e-08,
"cache_read_input_token_cost_flex": 1.5e-08,
"cache_read_input_token_cost_priority": 5e-08,
"input_cost_per_token": 3e-07,
"input_cost_per_token_batches": 1.5e-07,
@ -42027,7 +42027,7 @@
"vertex_ai/gemini-3.5-flash-lite": {
"deprecation_date": "2027-07-21",
"cache_read_input_token_cost": 3e-08,
"cache_read_input_token_cost_flex": 2e-08,
"cache_read_input_token_cost_flex": 1.5e-08,
"cache_read_input_token_cost_priority": 5e-08,
"input_cost_per_token": 3e-07,
"input_cost_per_token_batches": 1.5e-07,

View file

@ -20150,7 +20150,7 @@
"gemini-3.5-flash-lite": {
"deprecation_date": "2027-07-21",
"cache_read_input_token_cost": 3e-08,
"cache_read_input_token_cost_flex": 2e-08,
"cache_read_input_token_cost_flex": 1.5e-08,
"cache_read_input_token_cost_priority": 5e-08,
"input_cost_per_token": 3e-07,
"input_cost_per_token_batches": 1.5e-07,
@ -42027,7 +42027,7 @@
"vertex_ai/gemini-3.5-flash-lite": {
"deprecation_date": "2027-07-21",
"cache_read_input_token_cost": 3e-08,
"cache_read_input_token_cost_flex": 2e-08,
"cache_read_input_token_cost_flex": 1.5e-08,
"cache_read_input_token_cost_priority": 5e-08,
"input_cost_per_token": 3e-07,
"input_cost_per_token_batches": 1.5e-07,

View file

@ -3377,6 +3377,53 @@ def test_generic_cost_per_token_gemini_35_flash_lite(_local_model_cost_map):
assert completion_cost == pytest.approx(0.00125)
GEMINI_35_FLASH_LITE_TIER_RATES_BY_SURFACE = [
("gemini", None, 3e-07, 2.5e-06, 3e-08),
("gemini", "flex", 1.5e-07, 1.25e-06, 2e-08),
("gemini", "priority", 5.4e-07, 4.5e-06, 5e-08),
("vertex_ai", None, 3e-07, 2.5e-06, 3e-08),
("vertex_ai", "flex", 1.5e-07, 1.25e-06, 1.5e-08),
("vertex_ai", "priority", 5.4e-07, 4.5e-06, 5e-08),
]
@pytest.mark.parametrize(
"custom_llm_provider,service_tier,input_rate,output_rate,cache_read_rate",
GEMINI_35_FLASH_LITE_TIER_RATES_BY_SURFACE,
)
def test_gemini_35_flash_lite_service_tier_pricing(
custom_llm_provider, service_tier, input_rate, output_rate, cache_read_rate, _local_model_cost_map
):
"""Regression: Vertex publishes flash-lite flex context caching at $0.015/M while the
Gemini API publishes $0.02/M, so vertex_ai flex cache reads must bill 1.5e-08/token
instead of the 2e-08 the map used to carry, without disturbing the Gemini API rate."""
usage = Usage(
prompt_tokens=1_000,
completion_tokens=500,
total_tokens=1_500,
prompt_tokens_details=PromptTokensDetailsWrapper(cached_tokens=200, text_tokens=800),
)
prompt_cost, completion_cost = generic_cost_per_token(
model="gemini-3.5-flash-lite",
usage=usage,
custom_llm_provider=custom_llm_provider,
service_tier=service_tier,
)
assert prompt_cost == pytest.approx(800 * input_rate + 200 * cache_read_rate, rel=1e-9)
assert completion_cost == pytest.approx(500 * output_rate, rel=1e-9)
def test_gemini_35_flash_lite_flex_cache_read_map_entries(_local_model_cost_map):
"""Each map entry carries its own surface's published flex cache-read rate: the bare
and vertex_ai keys are the Vertex surface at $0.015/M, the gemini key is the Gemini
API surface at $0.02/M."""
assert litellm.model_cost["gemini-3.5-flash-lite"]["cache_read_input_token_cost_flex"] == 1.5e-08
assert litellm.model_cost["vertex_ai/gemini-3.5-flash-lite"]["cache_read_input_token_cost_flex"] == 1.5e-08
assert litellm.model_cost["gemini/gemini-3.5-flash-lite"]["cache_read_input_token_cost_flex"] == 2e-08
@pytest.mark.parametrize(
"service_tier,input_rate,cache_read_rate,cache_write_rate,output_rate",
[