litellm/tests/e2e/llm_translation/test_google_native_e2e.py
mubashir1osmani 4725cb4661
test(e2e): cover google-native generateContent framing and prometheus queue time (#34650)
* test(e2e): cover google-native generateContent framing and prometheus queue time

Adds live coverage for three shipped regressions that had none, all reached
through surfaces a customer drives from Google SDKs and operator dashboards.

The managed google-native route (`/v1beta/models/{model}:generateContent`) had
no harness support at all, so EndpointsClient gains generate_content and
stream_generate_content plus the request body models, and a new suite asserts
the two contracts that broke there: the response carries
x-litellm-response-cost so SDK traffic reconciles against spend (LIT-4076), and
the stream relays single-prefixed SSE frames with no OpenAI [DONE] terminator.
A doubled `data:` prefix, a leaked bytes literal, or the [DONE] sentinel each
fail the stream test; [DONE] absence is only asserted once real content has
arrived, because a first-chunk upstream error legitimately falls back to the
OpenAI error shape and does emit it.

The prometheus test pins litellm_request_queue_time_seconds to an actual
observation on our own key's series rather than to the family merely existing,
which is the distinction the original regression turned on: the histogram stayed
registered while nothing was ever written to it (LIT-2034).

Each assertion was mutation-checked against the live proxy; inverting the
[DONE] expectation, the cost-header expectation, or the metric name fails the
corresponding test.

* refactor(e2e): simplify google native coverage
2026-08-12 01:28:26 +00:00

107 lines
3.9 KiB
Python

from __future__ import annotations
import pytest
from pydantic import BaseModel
from e2e_config import unique_marker
from e2e_http import StreamingResponse, require_successful_call
from endpoints_client import EndpointsClient
from lifecycle import ResourceManager
from models import LiteLLMParamsBody
pytestmark = pytest.mark.e2e
UPSTREAM_MODEL = "gemini/gemini-2.5-flash"
class _StreamPart(BaseModel):
text: str | None = None
class _StreamContent(BaseModel):
parts: tuple[_StreamPart, ...] = ()
class _StreamCandidate(BaseModel):
content: _StreamContent | None = None
class _StreamEvent(BaseModel):
candidates: tuple[_StreamCandidate, ...] = ()
def _managed_deployment(client: EndpointsClient, resources: ResourceManager) -> str:
model = f"e2e-google-native-{unique_marker()}"
model_id = client.create_model(
model,
LiteLLMParamsBody(model=UPSTREAM_MODEL, api_key="os.environ/GEMINI_API_KEY"),
)
resources.defer(lambda: client.delete_model(model_id))
return model
def _streamed_text(result: StreamingResponse) -> str:
return "".join(
part.text
for event in result.stream_events
for candidate in _StreamEvent.model_validate_json(event).candidates
for part in (candidate.content.parts if candidate.content else ())
if part.text
)
class TestGoogleNativeGenerateContent:
@pytest.mark.covers("llm.google_native.gemini.basic.nonstream.cost_logged")
def test_generate_content_returns_response_cost_header(
self,
endpoints_client: EndpointsClient,
resources: ResourceManager,
scoped_key: str,
) -> None:
model = _managed_deployment(endpoints_client, resources)
result = endpoints_client.generate_content(
scoped_key, model, f"Reply with the single word ok. {unique_marker()}"
)
require_successful_call(result)
assert result.call_id, "generateContent must stamp x-litellm-call-id"
assert result.response_cost is not None, (
"generateContent returned no x-litellm-response-cost header; "
"google-native traffic cannot be reconciled against spend without it"
)
assert result.response_cost > 0, f"x-litellm-response-cost must be a real cost, got {result.response_cost}"
@pytest.mark.covers("llm.google_native.gemini.basic.stream.works")
def test_stream_generate_content_frames_sse_the_way_google_sdks_expect(
self,
endpoints_client: EndpointsClient,
resources: ResourceManager,
scoped_key: str,
) -> None:
model = _managed_deployment(endpoints_client, resources)
result = endpoints_client.generate_content(
scoped_key,
model,
f"Count from one to five, one number per line. {unique_marker()}",
stream=True,
)
require_successful_call(result)
assert result.is_streaming, f"expected text/event-stream, got content-type {result.content_type!r}"
assert result.stream_error is None, f"stream carried an error: {result.stream_error}"
assert result.stream_events, f"stream delivered no data events (chunks={result.chunks})"
doubled = tuple(event for event in result.stream_events if event.lstrip().startswith("data:"))
assert not doubled, (
f"{len(doubled)} event(s) carry a second data: prefix, so the proxy re-wrapped "
f"already-framed SSE; first offender: {doubled[0][:120]!r}"
)
leaked = tuple(event for event in result.stream_events if event.startswith("b'"))
assert not leaked, f"event serialized as a Python bytes literal instead of text: {leaked[0][:120]!r}"
assert _streamed_text(result).strip(), "stream delivered events but no candidate text"
assert not result.stream_done, (
"google-native stream emitted the OpenAI [DONE] sentinel; Google never sends it "
"and the Vertex Java SDK rejects the stream when it appears"
)