From 3a2728a42fe715b315c5c72e9ade106461680ad9 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Tue, 18 Aug 2026 21:18:16 -0700 Subject: [PATCH 1/2] test: derive vertex batch cost expectation from the cost map The gemini 3.6 flash batch rates landed at half the standard rates in 94a29e0708, so the hardcoded standard-rate expectation started failing on staging and red-lit misc / Run tests on every PR. --- tests/test_litellm/batches/test_batch_utils.py | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/tests/test_litellm/batches/test_batch_utils.py b/tests/test_litellm/batches/test_batch_utils.py index 573882ebfca..49ff8de1ab9 100644 --- a/tests/test_litellm/batches/test_batch_utils.py +++ b/tests/test_litellm/batches/test_batch_utils.py @@ -890,8 +890,13 @@ async def test_handle_completed_vertex_batch_computes_cost_usage_and_models(monk litellm_params={"vertex_project": "proj-1", "vertex_location": "us-central1"}, ) + pricing = litellm.model_cost["vertex_ai/gemini-3.6-flash"] + batch_input = pricing["input_cost_per_token_batches"] + batch_output = pricing["output_cost_per_token_batches"] + + assert batch_input < pricing["input_cost_per_token"] assert cost > 0 - assert cost == pytest.approx(30 * 7.5e-07 + 15 * 3.75e-06) + assert cost == pytest.approx(30 * batch_input + 15 * batch_output) assert (usage.prompt_tokens, usage.completion_tokens, usage.total_tokens) == (30, 15, 45) assert models == ["gemini-3.6-flash", "gemini-3.6-flash"] From df0d8ff15ea2f65e9f1ba90209239f241cf16d91 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Tue, 18 Aug 2026 22:38:13 -0700 Subject: [PATCH 2/2] test: assert the vertex batch output rate is a real discount --- tests/test_litellm/batches/test_batch_utils.py | 1 + 1 file changed, 1 insertion(+) diff --git a/tests/test_litellm/batches/test_batch_utils.py b/tests/test_litellm/batches/test_batch_utils.py index 49ff8de1ab9..50f1380a4f8 100644 --- a/tests/test_litellm/batches/test_batch_utils.py +++ b/tests/test_litellm/batches/test_batch_utils.py @@ -895,6 +895,7 @@ async def test_handle_completed_vertex_batch_computes_cost_usage_and_models(monk batch_output = pricing["output_cost_per_token_batches"] assert batch_input < pricing["input_cost_per_token"] + assert batch_output < pricing["output_cost_per_token"] assert cost > 0 assert cost == pytest.approx(30 * batch_input + 15 * batch_output) assert (usage.prompt_tokens, usage.completion_tokens, usage.total_tokens) == (30, 15, 45)