From b00c00e5f83daa8b2598511b31daa8d8331bb48a Mon Sep 17 00:00:00 2001 From: Ankit Jha Date: Fri, 11 Sep 2026 15:49:20 +0530 Subject: [PATCH 1/7] fix: record vertex_location on the generate_content path so cost uses the configured region The Vertex location decides the regional pricing uplift. The completion path sets it on litellm_params and async_anthropic_messages_handler records it on the logging object, but generate_content passed only litellm_call_id, so the cost calculator never saw it. The result is that a model configured `vertex_location: global` is priced as us-central1 on POST /v1beta/models/{model}:generateContent and costs 10% more than the same model over /v1/chat/completions. Two requests differing only in the URL are billed differently, and spend cannot be reconciled against the cloud bill. Uses the same VertexBase.explicit_vertex_ai_location helper the anthropic messages path already uses, so an unset location records nothing and the cost calculator keeps its own fallback. Fixes #40692 Signed-off-by: Ankit Jha --- litellm/google_genai/main.py | 15 ++++ .../google_genai/test_google_genai_main.py | 70 +++++++++++++++++++ 2 files changed, 85 insertions(+) diff --git a/litellm/google_genai/main.py b/litellm/google_genai/main.py index c1822e4720d..627c2919fd4 100644 --- a/litellm/google_genai/main.py +++ b/litellm/google_genai/main.py @@ -2,6 +2,7 @@ import asyncio import contextvars from collections.abc import Iterator from functools import partial +from types import MappingProxyType from typing import TYPE_CHECKING, Any, ClassVar, Final import httpx @@ -192,12 +193,26 @@ class GenerateContentHelper: if litellm_logging_obj is None: raise ValueError("litellm_logging_obj is required, but got None") + # The configured Vertex location decides the regional pricing uplift. The + # completion and anthropic_messages paths both record it here; without it + # the cost calculator falls back to a default region and prices a `global` + # model as us-central1, so the same model costs 10% more on this route. + from litellm.llms.vertex_ai.vertex_llm_base import VertexBase + + explicit_vertex_location: Final = VertexBase.explicit_vertex_ai_location( + MappingProxyType(litellm_params.model_dump(exclude_none=True)) + ) + vertex_location_params: Final = ( + {"vertex_location": explicit_vertex_location} if explicit_vertex_location else {} + ) + litellm_logging_obj.update_from_kwargs( kwargs=kwargs, model=model, optional_params=dict(generate_content_config_dict), litellm_params={ "litellm_call_id": litellm_call_id, + **vertex_location_params, }, custom_llm_provider=custom_llm_provider, ) diff --git a/tests/test_litellm/google_genai/test_google_genai_main.py b/tests/test_litellm/google_genai/test_google_genai_main.py index 238fff7deca..eef7ae02a4b 100644 --- a/tests/test_litellm/google_genai/test_google_genai_main.py +++ b/tests/test_litellm/google_genai/test_google_genai_main.py @@ -259,3 +259,73 @@ async def test_native_fields_forwarded_on_async_stream(): body = mock_post.call_args.kwargs["json"] assert body["safetySettings"] == safety_settings assert "safetySettings" not in body.get("generationConfig", {}) + + +def _generate_content_logging_obj(call_id: str): + from datetime import datetime + + from litellm.litellm_core_utils.litellm_logging import Logging + + return Logging( + model="gemini-2.0-flash", + messages=[], + stream=False, + call_type="generate_content", + start_time=datetime.now(), + litellm_call_id=call_id, + function_id=call_id, + ) + + +@pytest.mark.parametrize("configured_location", ["global", "us-east5"]) +def test_vertex_location_recorded_for_cost_calculation(configured_location): + """ + Regression for https://github.com/BerriAI/litellm/issues/40692 + + The Vertex location decides the regional pricing uplift. The completion and + anthropic_messages paths record it on the logging object so the cost + calculator can price against the configured region. generate_content did + not, so a model configured `vertex_location: global` was priced as + us-central1 and cost 10% more than the same model over /v1/chat/completions. + """ + from litellm.google_genai.main import GenerateContentHelper + + logging_obj = _generate_content_logging_obj("vertex-location-test") + + GenerateContentHelper.setup_generate_content_call( + model="vertex_ai/gemini-2.0-flash", + contents=[{"role": "user", "parts": [{"text": "say ok"}]}], + custom_llm_provider="vertex_ai", + litellm_logging_obj=logging_obj, + litellm_call_id="vertex-location-test", + vertex_project="test-project", + vertex_location=configured_location, + ) + + recorded = logging_obj.model_call_details["litellm_params"] + assert recorded.get("vertex_location") == configured_location, ( + "the configured vertex_location must reach the cost calculator; " + f"got {recorded.get('vertex_location')!r}" + ) + + +def test_vertex_location_absent_when_not_configured(): + """ + Nothing configured means nothing recorded, so the cost calculator keeps its + own fallback rather than being handed an empty value here. + """ + from litellm.google_genai.main import GenerateContentHelper + + logging_obj = _generate_content_logging_obj("vertex-location-unset") + + GenerateContentHelper.setup_generate_content_call( + model="gemini/gemini-2.0-flash", + contents=[{"role": "user", "parts": [{"text": "say ok"}]}], + custom_llm_provider="gemini", + litellm_logging_obj=logging_obj, + litellm_call_id="vertex-location-unset", + api_key="test-key", + ) + + recorded = logging_obj.model_call_details["litellm_params"] + assert "vertex_location" not in recorded From 674b95e12164f84043528fbd169a798769c20e9b Mon Sep 17 00:00:00 2001 From: Ankit Jha Date: Wed, 16 Sep 2026 15:15:24 +0530 Subject: [PATCH 2/7] fix: satisfy the LIT002 type-discipline gate for vertex_location_params Each ternary branch now wraps its own dict literal in MappingProxyType, so both count as an argument passed directly to a freezing wrapper instead of a bare mutable-collection construction. Confirmed against the actual checker: litellm/ scores identically to the merge base with this change applied (26719 either way). Signed-off-by: Ankit Jha --- litellm/google_genai/main.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/litellm/google_genai/main.py b/litellm/google_genai/main.py index 627c2919fd4..3daf43c729e 100644 --- a/litellm/google_genai/main.py +++ b/litellm/google_genai/main.py @@ -203,7 +203,9 @@ class GenerateContentHelper: MappingProxyType(litellm_params.model_dump(exclude_none=True)) ) vertex_location_params: Final = ( - {"vertex_location": explicit_vertex_location} if explicit_vertex_location else {} + MappingProxyType({"vertex_location": explicit_vertex_location}) + if explicit_vertex_location + else MappingProxyType({}) ) litellm_logging_obj.update_from_kwargs( From 8c2f3616cba61546fe86a5e17e022e053dd88fa8 Mon Sep 17 00:00:00 2001 From: Ankit Jha Date: Thu, 1 Oct 2026 17:29:26 +0530 Subject: [PATCH 3/7] test: keep the vertex_location assertion message on one line for ruff format Signed-off-by: Ankit Jha --- tests/unit/google_genai/test_google_genai_main.py | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/tests/unit/google_genai/test_google_genai_main.py b/tests/unit/google_genai/test_google_genai_main.py index eef7ae02a4b..67fef86657d 100644 --- a/tests/unit/google_genai/test_google_genai_main.py +++ b/tests/unit/google_genai/test_google_genai_main.py @@ -304,8 +304,7 @@ def test_vertex_location_recorded_for_cost_calculation(configured_location): recorded = logging_obj.model_call_details["litellm_params"] assert recorded.get("vertex_location") == configured_location, ( - "the configured vertex_location must reach the cost calculator; " - f"got {recorded.get('vertex_location')!r}" + f"the configured vertex_location must reach the cost calculator; got {recorded.get('vertex_location')!r}" ) From b67dbfa9480c7307723cfe6416fae88c7ff69794 Mon Sep 17 00:00:00 2001 From: Ankit Jha Date: Thu, 1 Oct 2026 18:27:18 +0530 Subject: [PATCH 4/7] test(google_genai): use a fixed timestamp in the logging helper --- tests/unit/google_genai/test_google_genai_main.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/unit/google_genai/test_google_genai_main.py b/tests/unit/google_genai/test_google_genai_main.py index 67fef86657d..497190ff914 100644 --- a/tests/unit/google_genai/test_google_genai_main.py +++ b/tests/unit/google_genai/test_google_genai_main.py @@ -271,7 +271,7 @@ def _generate_content_logging_obj(call_id: str): messages=[], stream=False, call_type="generate_content", - start_time=datetime.now(), + start_time=datetime(2025, 1, 1), litellm_call_id=call_id, function_id=call_id, ) From 9d40dc7b264f0d517b61043977bb19b55874e372 Mon Sep 17 00:00:00 2001 From: Ankit Jha Date: Thu, 1 Oct 2026 19:42:56 +0530 Subject: [PATCH 5/7] perf(google_genai): read only the two location keys instead of dumping every param --- litellm/google_genai/main.py | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/litellm/google_genai/main.py b/litellm/google_genai/main.py index 3daf43c729e..86f98d856b7 100644 --- a/litellm/google_genai/main.py +++ b/litellm/google_genai/main.py @@ -200,7 +200,12 @@ class GenerateContentHelper: from litellm.llms.vertex_ai.vertex_llm_base import VertexBase explicit_vertex_location: Final = VertexBase.explicit_vertex_ai_location( - MappingProxyType(litellm_params.model_dump(exclude_none=True)) + MappingProxyType( + { + key: getattr(litellm_params, key, None) + for key in ("vertex_location", "vertex_ai_location") + } + ) ) vertex_location_params: Final = ( MappingProxyType({"vertex_location": explicit_vertex_location}) From 849baa8f8fbde498bf588dda52861dc7d39ab382 Mon Sep 17 00:00:00 2001 From: Ankit Jha Date: Thu, 1 Oct 2026 21:27:28 +0530 Subject: [PATCH 6/7] chore(google_genai): shorten comments --- litellm/google_genai/main.py | 6 ++---- tests/unit/google_genai/test_google_genai_main.py | 6 +----- 2 files changed, 3 insertions(+), 9 deletions(-) diff --git a/litellm/google_genai/main.py b/litellm/google_genai/main.py index 86f98d856b7..db1a11341fb 100644 --- a/litellm/google_genai/main.py +++ b/litellm/google_genai/main.py @@ -193,10 +193,8 @@ class GenerateContentHelper: if litellm_logging_obj is None: raise ValueError("litellm_logging_obj is required, but got None") - # The configured Vertex location decides the regional pricing uplift. The - # completion and anthropic_messages paths both record it here; without it - # the cost calculator falls back to a default region and prices a `global` - # model as us-central1, so the same model costs 10% more on this route. + # Record the configured location so a `global` model is not priced as + # us-central1 (+10%), matching the completion path. from litellm.llms.vertex_ai.vertex_llm_base import VertexBase explicit_vertex_location: Final = VertexBase.explicit_vertex_ai_location( diff --git a/tests/unit/google_genai/test_google_genai_main.py b/tests/unit/google_genai/test_google_genai_main.py index 497190ff914..275264e2542 100644 --- a/tests/unit/google_genai/test_google_genai_main.py +++ b/tests/unit/google_genai/test_google_genai_main.py @@ -282,11 +282,7 @@ def test_vertex_location_recorded_for_cost_calculation(configured_location): """ Regression for https://github.com/BerriAI/litellm/issues/40692 - The Vertex location decides the regional pricing uplift. The completion and - anthropic_messages paths record it on the logging object so the cost - calculator can price against the configured region. generate_content did - not, so a model configured `vertex_location: global` was priced as - us-central1 and cost 10% more than the same model over /v1/chat/completions. + generate_content must record vertex_location so `global` is not priced as us-central1. """ from litellm.google_genai.main import GenerateContentHelper From 162d60cf387c680db48d526d18f00ffd242d9e45 Mon Sep 17 00:00:00 2001 From: Ankit Jha Date: Thu, 1 Oct 2026 21:28:14 +0530 Subject: [PATCH 7/7] style(google_genai): ruff format and tz-aware fixed timestamp --- litellm/google_genai/main.py | 5 +---- tests/unit/google_genai/test_google_genai_main.py | 4 ++-- 2 files changed, 3 insertions(+), 6 deletions(-) diff --git a/litellm/google_genai/main.py b/litellm/google_genai/main.py index db1a11341fb..25d7de9f710 100644 --- a/litellm/google_genai/main.py +++ b/litellm/google_genai/main.py @@ -199,10 +199,7 @@ class GenerateContentHelper: explicit_vertex_location: Final = VertexBase.explicit_vertex_ai_location( MappingProxyType( - { - key: getattr(litellm_params, key, None) - for key in ("vertex_location", "vertex_ai_location") - } + {key: getattr(litellm_params, key, None) for key in ("vertex_location", "vertex_ai_location")} ) ) vertex_location_params: Final = ( diff --git a/tests/unit/google_genai/test_google_genai_main.py b/tests/unit/google_genai/test_google_genai_main.py index 275264e2542..31e2f56092c 100644 --- a/tests/unit/google_genai/test_google_genai_main.py +++ b/tests/unit/google_genai/test_google_genai_main.py @@ -262,7 +262,7 @@ async def test_native_fields_forwarded_on_async_stream(): def _generate_content_logging_obj(call_id: str): - from datetime import datetime + from datetime import datetime, timezone from litellm.litellm_core_utils.litellm_logging import Logging @@ -271,7 +271,7 @@ def _generate_content_logging_obj(call_id: str): messages=[], stream=False, call_type="generate_content", - start_time=datetime(2025, 1, 1), + start_time=datetime(2025, 1, 1, tzinfo=timezone.utc), litellm_call_id=call_id, function_id=call_id, )