From c71b6fffa3b088d3824b2b2344b6679a538a3cac Mon Sep 17 00:00:00 2001 From: Kris Xia Date: Mon, 11 May 2026 14:11:25 +0800 Subject: [PATCH] chore(vertex_ai/gemini): translate code comments to english Translate the Chinese comments and docstrings introduced in this PR to English to match the repository convention for upstream BerriAI/litellm. No behavior change. --- .../llms/vertex_ai/gemini/transformation.py | 36 ++++++++++++------- .../llms/vertex_ai/test_vertex.py | 32 +++++++++-------- 2 files changed, 41 insertions(+), 27 deletions(-) diff --git a/litellm/llms/vertex_ai/gemini/transformation.py b/litellm/llms/vertex_ai/gemini/transformation.py index 07eb03008d2..15e69358b9e 100644 --- a/litellm/llms/vertex_ai/gemini/transformation.py +++ b/litellm/llms/vertex_ai/gemini/transformation.py @@ -60,8 +60,9 @@ from ..common_utils import ( get_supports_system_message, ) -# 使用 Any 标注避免在模块加载期引入到 vertex_llm_base 的循环依赖; -# 该实例在首次需要拉取 GCS metadata 时才通过 _get_vertex_base() 懒加载。 +# Typed as Any to avoid introducing a module-load-time cyclic import to +# vertex_llm_base. The instance is lazily constructed by _get_vertex_base() +# the first time GCS metadata needs to be fetched. _GCS_METADATA_VERTEX_BASE: Optional[Any] = None _GEMINI_MIME_TYPE_ALIASES: Dict[str, str] = { "image/jpg": "image/jpeg", @@ -69,7 +70,7 @@ _GEMINI_MIME_TYPE_ALIASES: Dict[str, str] = { def _get_vertex_base() -> Any: - """懒加载共享的 VertexBase 实例,避免模块加载期的循环依赖。""" + """Lazily return the shared VertexBase instance to avoid a module-load-time cyclic import.""" global _GCS_METADATA_VERTEX_BASE if _GCS_METADATA_VERTEX_BASE is None: from ..vertex_llm_base import VertexBase @@ -231,10 +232,12 @@ def _get_gcs_object_content_type( """ Resolve content type from GCS object metadata. - 仅当调用方显式传入 Vertex 凭据时才附带 Bearer token,避免在 Gemini - API key(Google AI Studio)路径上自动使用服务端的默认 Google 凭据去 - 访问 GCS,防止被用作探测私有 GCS 对象的 oracle。 - 无显式凭据时只做匿名请求,仅对公开可读对象有效。 + Only attaches a Bearer token when the caller explicitly supplies Vertex + credentials, to avoid using the server's default Google credentials on + the Gemini API-key (Google AI Studio) path and being used as an oracle + for private GCS object metadata. Without explicit credentials we only + issue an anonymous request, which only succeeds for publicly-readable + objects. """ try: bucket, object_name = _parse_gs_uri(image_url) @@ -264,8 +267,9 @@ def _get_gcs_object_content_type( llm_provider="vertex_ai", ) - # 通过 httpx.URL 固定 scheme/host,并对 bucket、object 都做 URL 编码, - # 避免被 CodeQL 误判为可能拼接出任意主机 URL 的 SSRF。 + # Build the URL via httpx.URL with a fixed scheme/host and URL-encode both + # bucket and object so CodeQL does not flag the interpolation as a + # potential SSRF that could resolve to an arbitrary host. encoded_bucket = quote(bucket, safe="") encoded_object = quote(object_name, safe="") metadata_url = httpx.URL( @@ -288,7 +292,8 @@ def _get_gcs_object_content_type( def _normalize_and_validate_gemini_mime_type( mime_type: str, model: Optional[str] ) -> str: - # 延迟导入,避免在模块顶层与 litellm.types.files 形成循环引用告警。 + # Import lazily to avoid a module-level cyclic-import alert with + # litellm.types.files. from litellm.types.files import get_file_extension_from_mime_type normalized_mime_type = _GEMINI_MIME_TYPE_ALIASES.get( @@ -555,6 +560,8 @@ def _gemini_convert_messages_with_history( # noqa: PLR0915 model=model, llm_provider="vertex_ai", ) + # TypedDict does not declare mime_type/content_type; + # read via Dict[str, Any] for caller-provided MIME fields. image_url_dict = cast(Dict[str, Any], raw_image_url) format = ( image_url_dict.get("format") @@ -612,6 +619,8 @@ def _gemini_convert_messages_with_history( # noqa: PLR0915 model=model, llm_provider="vertex_ai", ) + # TypedDict does not declare mime_type/content_type; + # read via Dict[str, Any] for caller-provided MIME fields. file_dict = cast(Dict[str, Any], _file_field) file_id = file_dict.get("file_id") format = ( @@ -1128,9 +1137,10 @@ async def async_transform_request_body( vertex_auth_header=vertex_auth_header, ) - # _transform_request_body 可能通过 _get_gcs_object_content_type 发起同步 httpx.get - # (最长 5s 超时)去拉 GCS 对象 metadata。为避免阻塞 async 事件循环,整个同步 - # 转换放到 worker 线程执行。 + # _transform_request_body may issue a sync httpx.get (up to 5s timeout) + # via _get_gcs_object_content_type to fetch GCS object metadata. Run the + # whole sync transformation on a worker thread so it does not block the + # async event loop. return await asyncify(_transform_request_body)( messages=messages, model=model, diff --git a/tests/test_litellm/llms/vertex_ai/test_vertex.py b/tests/test_litellm/llms/vertex_ai/test_vertex.py index c24b03e8c71..c5a676168ae 100644 --- a/tests/test_litellm/llms/vertex_ai/test_vertex.py +++ b/tests/test_litellm/llms/vertex_ai/test_vertex.py @@ -1283,7 +1283,7 @@ def test_process_gemini_media(): def test_process_gemini_media_gcs_without_extension_raises_clear_error(): - # mock 掉 GCS metadata 查询,避免测试发出真实外网请求 + # Mock the GCS metadata lookup to avoid real outbound HTTP in tests. with patch( "litellm.llms.vertex_ai.gemini.transformation._get_gcs_object_content_type", return_value=None, @@ -1459,10 +1459,11 @@ def test_get_gcs_object_content_type_fails_fast_with_explicit_credentials(): def test_get_gcs_object_content_type_without_credentials_skips_auth(): - """未显式提供 Vertex 凭据时,不应使用服务端默认凭据去访问 GCS。 + """Without explicit Vertex credentials, must not use the server's default + Google credentials to access GCS. - 这是为了避免 Gemini API key(Google AI Studio)路径被用作探测私有 - GCS 对象的 oracle(veria-ai 审阅反馈)。 + Prevents the Gemini API-key (Google AI Studio) path from being used as an + oracle for private GCS objects (per veria-ai review feedback). """ from litellm.llms.vertex_ai.gemini import transformation as gemini_transformation @@ -1484,9 +1485,9 @@ def test_get_gcs_object_content_type_without_credentials_skips_auth(): image_url="gs://public-bucket/public-object" ) - # 不应触发 get_access_token(避免使用服务端默认凭据) + # Must not call get_access_token (so default server credentials are not used) mock_vertex_base.get_access_token.assert_not_called() - # 匿名请求仍会发出,用于公开可读对象 + # An anonymous request is still sent, covering publicly-readable objects. mock_http_get.assert_called_once() call_kwargs = mock_http_get.call_args.kwargs assert "Authorization" not in call_kwargs.get("headers", {}) @@ -1494,7 +1495,7 @@ def test_get_gcs_object_content_type_without_credentials_skips_auth(): def test_async_transform_request_body_does_not_block_event_loop(): - """当同步 GCS metadata 查询阻塞时,async_transform_request_body 不应阻塞事件循环。""" + """When the sync GCS metadata lookup blocks, async_transform_request_body must not block the event loop.""" import asyncio import time @@ -1512,9 +1513,10 @@ def test_async_transform_request_body_does_not_block_event_loop(): } ] - # 模拟同步阻塞的 httpx.get(就像真实的 GCS metadata 查询超时), - # 如果 async_transform_request_body 没有把同步部分 offload 到 worker 线程, - # 这次阻塞会卡住事件循环,并行的 sleep 就无法推进。 + # Simulate a blocking sync httpx.get (as if the real GCS metadata lookup + # timed out). If async_transform_request_body does not offload the sync + # portion to a worker thread, this blocks the event loop and a concurrent + # sleep cannot make progress. def slow_http_get(*args, **kwargs): time.sleep(0.5) response = MagicMock() @@ -1522,7 +1524,7 @@ def test_async_transform_request_body_does_not_block_event_loop(): response.json.return_value = {"contentType": "image/png"} return response - # 模拟 context-caching 查询,避免其发出真实 HTTP + # Stub out the context-caching lookup so it does not make a real HTTP call. async def fake_check_and_create_cache(self, **kwargs): return kwargs["messages"], kwargs["optional_params"], None @@ -1570,10 +1572,12 @@ def test_async_transform_request_body_does_not_block_event_loop(): ): sleep_elapsed = asyncio.run(run_scenario()) - # 没被阻塞的情况下,0.05s 的 sleep 不应被拖到接近同步阻塞时长(0.5s) + # If the loop is not blocked, a 0.05s sleep must not be stretched toward + # the sync block duration (0.5s). assert sleep_elapsed < 0.4, ( - f"事件循环被阻塞 {sleep_elapsed:.3f}s,async_transform_request_body 未把同步 GCS " - "metadata 查询 offload 到 worker 线程" + f"Event loop blocked for {sleep_elapsed:.3f}s; " + "async_transform_request_body did not offload the sync GCS metadata " + "lookup to a worker thread" )