chore(vertex_ai/gemini): translate code comments to english

Translate the Chinese comments and docstrings introduced in this PR to
English to match the repository convention for upstream BerriAI/litellm.
No behavior change.
This commit is contained in:
Kris Xia 2026-05-11 14:11:25 +08:00 • committed by S0ngRu1
parent d0061fbea7
commit c71b6fffa3
2 changed files with 41 additions and 27 deletions

View file

@ -60,8 +60,9 @@ from ..common_utils import (
get_supports_system_message,
)
# 使用 Any 标注避免在模块加载期引入到 vertex_llm_base 的循环依赖;
# 该实例在首次需要拉取 GCS metadata 时才通过 _get_vertex_base() 懒加载。
# Typed as Any to avoid introducing a module-load-time cyclic import to
# vertex_llm_base. The instance is lazily constructed by _get_vertex_base()
# the first time GCS metadata needs to be fetched.
_GCS_METADATA_VERTEX_BASE: Optional[Any] = None
_GEMINI_MIME_TYPE_ALIASES: Dict[str, str] = {
"image/jpg": "image/jpeg",
@ -69,7 +70,7 @@ _GEMINI_MIME_TYPE_ALIASES: Dict[str, str] = {
def _get_vertex_base() -> Any:
"""懒加载共享的 VertexBase 实例,避免模块加载期的循环依赖。"""
"""Lazily return the shared VertexBase instance to avoid a module-load-time cyclic import."""
global _GCS_METADATA_VERTEX_BASE
if _GCS_METADATA_VERTEX_BASE is None:
from ..vertex_llm_base import VertexBase
@ -231,10 +232,12 @@ def _get_gcs_object_content_type(
"""
Resolve content type from GCS object metadata.
仅当调用方显式传入 Vertex 凭据时才附带 Bearer token,避免在 Gemini
API key(Google AI Studio)路径上自动使用服务端的默认 Google 凭据去
访问 GCS,防止被用作探测私有 GCS 对象的 oracle。
无显式凭据时只做匿名请求,仅对公开可读对象有效。
Only attaches a Bearer token when the caller explicitly supplies Vertex
credentials, to avoid using the server's default Google credentials on
the Gemini API-key (Google AI Studio) path and being used as an oracle
for private GCS object metadata. Without explicit credentials we only
issue an anonymous request, which only succeeds for publicly-readable
objects.
"""
try:
bucket, object_name = _parse_gs_uri(image_url)
@ -264,8 +267,9 @@ def _get_gcs_object_content_type(
llm_provider="vertex_ai",
)
# 通过 httpx.URL 固定 scheme/host,并对 bucket、object 都做 URL 编码,
# 避免被 CodeQL 误判为可能拼接出任意主机 URL 的 SSRF。
# Build the URL via httpx.URL with a fixed scheme/host and URL-encode both
# bucket and object so CodeQL does not flag the interpolation as a
# potential SSRF that could resolve to an arbitrary host.
encoded_bucket = quote(bucket, safe="")
encoded_object = quote(object_name, safe="")
metadata_url = httpx.URL(
@ -288,7 +292,8 @@ def _get_gcs_object_content_type(
def _normalize_and_validate_gemini_mime_type(
mime_type: str, model: Optional[str]
) -> str:
# 延迟导入,避免在模块顶层与 litellm.types.files 形成循环引用告警。
# Import lazily to avoid a module-level cyclic-import alert with
# litellm.types.files.
from litellm.types.files import get_file_extension_from_mime_type
normalized_mime_type = _GEMINI_MIME_TYPE_ALIASES.get(
@ -555,6 +560,8 @@ def _gemini_convert_messages_with_history( # noqa: PLR0915
model=model,
llm_provider="vertex_ai",
)
# TypedDict does not declare mime_type/content_type;
# read via Dict[str, Any] for caller-provided MIME fields.
image_url_dict = cast(Dict[str, Any], raw_image_url)
format = (
image_url_dict.get("format")
@ -612,6 +619,8 @@ def _gemini_convert_messages_with_history( # noqa: PLR0915
model=model,
llm_provider="vertex_ai",
)
# TypedDict does not declare mime_type/content_type;
# read via Dict[str, Any] for caller-provided MIME fields.
file_dict = cast(Dict[str, Any], _file_field)
file_id = file_dict.get("file_id")
format = (
@ -1128,9 +1137,10 @@ async def async_transform_request_body(
vertex_auth_header=vertex_auth_header,
)
# _transform_request_body 可能通过 _get_gcs_object_content_type 发起同步 httpx.get
# (最长 5s 超时)去拉 GCS 对象 metadata。为避免阻塞 async 事件循环,整个同步
# 转换放到 worker 线程执行。
# _transform_request_body may issue a sync httpx.get (up to 5s timeout)
# via _get_gcs_object_content_type to fetch GCS object metadata. Run the
# whole sync transformation on a worker thread so it does not block the
# async event loop.
return await asyncify(_transform_request_body)(
messages=messages,
model=model,

View file

@ -1283,7 +1283,7 @@ def test_process_gemini_media():
def test_process_gemini_media_gcs_without_extension_raises_clear_error():
# mock 掉 GCS metadata 查询,避免测试发出真实外网请求
# Mock the GCS metadata lookup to avoid real outbound HTTP in tests.
with patch(
"litellm.llms.vertex_ai.gemini.transformation._get_gcs_object_content_type",
return_value=None,
@ -1459,10 +1459,11 @@ def test_get_gcs_object_content_type_fails_fast_with_explicit_credentials():
def test_get_gcs_object_content_type_without_credentials_skips_auth():
"""未显式提供 Vertex 凭据时,不应使用服务端默认凭据去访问 GCS。
"""Without explicit Vertex credentials, must not use the server's default
Google credentials to access GCS.
这是为了避免 Gemini API key(Google AI Studio)路径被用作探测私有
GCS 对象的 oracle(veria-ai 审阅反馈)。
Prevents the Gemini API-key (Google AI Studio) path from being used as an
oracle for private GCS objects (per veria-ai review feedback).
"""
from litellm.llms.vertex_ai.gemini import transformation as gemini_transformation
@ -1484,9 +1485,9 @@ def test_get_gcs_object_content_type_without_credentials_skips_auth():
image_url="gs://public-bucket/public-object"
)
# 不应触发 get_access_token(避免使用服务端默认凭据)
# Must not call get_access_token (so default server credentials are not used)
mock_vertex_base.get_access_token.assert_not_called()
# 匿名请求仍会发出,用于公开可读对象
# An anonymous request is still sent, covering publicly-readable objects.
mock_http_get.assert_called_once()
call_kwargs = mock_http_get.call_args.kwargs
assert "Authorization" not in call_kwargs.get("headers", {})
@ -1494,7 +1495,7 @@ def test_get_gcs_object_content_type_without_credentials_skips_auth():
def test_async_transform_request_body_does_not_block_event_loop():
"""当同步 GCS metadata 查询阻塞时,async_transform_request_body 不应阻塞事件循环。"""
"""When the sync GCS metadata lookup blocks, async_transform_request_body must not block the event loop."""
import asyncio
import time
@ -1512,9 +1513,10 @@ def test_async_transform_request_body_does_not_block_event_loop():
}
]
# 模拟同步阻塞的 httpx.get(就像真实的 GCS metadata 查询超时),
# 如果 async_transform_request_body 没有把同步部分 offload 到 worker 线程,
# 这次阻塞会卡住事件循环,并行的 sleep 就无法推进。
# Simulate a blocking sync httpx.get (as if the real GCS metadata lookup
# timed out). If async_transform_request_body does not offload the sync
# portion to a worker thread, this blocks the event loop and a concurrent
# sleep cannot make progress.
def slow_http_get(*args, **kwargs):
time.sleep(0.5)
response = MagicMock()
@ -1522,7 +1524,7 @@ def test_async_transform_request_body_does_not_block_event_loop():
response.json.return_value = {"contentType": "image/png"}
return response
# 模拟 context-caching 查询,避免其发出真实 HTTP
# Stub out the context-caching lookup so it does not make a real HTTP call.
async def fake_check_and_create_cache(self, **kwargs):
return kwargs["messages"], kwargs["optional_params"], None
@ -1570,10 +1572,12 @@ def test_async_transform_request_body_does_not_block_event_loop():
):
sleep_elapsed = asyncio.run(run_scenario())
# 没被阻塞的情况下,0.05s 的 sleep 不应被拖到接近同步阻塞时长(0.5s)
# If the loop is not blocked, a 0.05s sleep must not be stretched toward
# the sync block duration (0.5s).
assert sleep_elapsed < 0.4, (
f"事件循环被阻塞 {sleep_elapsed:.3f}s,async_transform_request_body 未把同步 GCS "
"metadata 查询 offload 到 worker 线程"
f"Event loop blocked for {sleep_elapsed:.3f}s; "
"async_transform_request_body did not offload the sync GCS metadata "
"lookup to a worker thread"
)