test(e2e): require spent completion tokens before accepting empty fallback content

The first pass accepted any empty completion whose finish_reason was "length",
which also swallowed a fallback that produced nothing at all. Require the
response to have billed completion tokens as well, so empty content is
accepted only when the budget was demonstrably spent on non-visible reasoning.

Asserts on completion_tokens rather than reasoning_tokens because the latter
is provider-optional; with empty content and a refusal of null, consumed
completion tokens are reasoning by elimination, since visible text would be
content. Both counts are reported in the failure message.

Folds the three body accessors onto one _parsed helper instead of re-parsing
per call, and adds completion_tokens_of / reasoning_tokens_of alongside.
This commit is contained in:
Yuneng Jiang 2026-08-29 16:19:41 -07:00
parent 6e341b79d8
commit fd4b540a44
No known key found for this signature in database
2 changed files with 38 additions and 17 deletions

View file

@ -66,24 +66,39 @@ def chat_override(
)
def _parsed(resp: StreamingResponse) -> ChatResponse | None:
try:
return ChatResponse.model_validate_json(resp.body)
except ValidationError:
return None
def content_of(resp: StreamingResponse) -> str | None:
"""The assistant message content of a successful chat response, or None when the
body is not a success shape (an error body, or an elided streamed body)."""
try:
parsed = ChatResponse.model_validate_json(resp.body)
except ValidationError:
return None
if not parsed.choices:
parsed = _parsed(resp)
if parsed is None or not parsed.choices:
return None
message = parsed.choices[0].message
return message.content if message is not None else None
def finish_reason_of(resp: StreamingResponse) -> str | None:
try:
parsed = ChatResponse.model_validate_json(resp.body)
except ValidationError:
return None
if not parsed.choices:
parsed = _parsed(resp)
if parsed is None or not parsed.choices:
return None
return parsed.choices[0].finish_reason
def completion_tokens_of(resp: StreamingResponse) -> int | None:
parsed = _parsed(resp)
if parsed is None or parsed.usage is None:
return None
return parsed.usage.completion_tokens
def reasoning_tokens_of(resp: StreamingResponse) -> int | None:
parsed = _parsed(resp)
if parsed is None or parsed.usage is None or parsed.usage.completion_tokens_details is None:
return None
return parsed.usage.completion_tokens_details.reasoning_tokens

View file

@ -5,9 +5,10 @@ Each test registers a primary deployment that fails (an unreachable base URL, or
a 1ms deadline) and calls it with a `router_settings_override` mapping it to the
real `gpt-5.5`. The proof the fallback fired is twofold: the response is a
completion from `gpt-5.5`, and the proxy reports at least one attempted fallback
in the x-litellm-attempted-fallbacks header. Empty content is accepted only for
`finish_reason == "length"`, since gpt-5.5 counts reasoning against max_tokens
and can consume the whole budget before emitting any text.
in the x-litellm-attempted-fallbacks header. Empty content is accepted only when
`finish_reason == "length"` and the response billed completion tokens, since
gpt-5.5 counts reasoning against max_tokens and can consume the whole budget
before emitting any text; a fallback that produced nothing at all still fails.
"""
from __future__ import annotations
@ -21,10 +22,12 @@ from lifecycle import ResourceManager
from models import RouterSettingsOverride
from reliability_support import (
chat_override,
completion_tokens_of,
content_of,
create_bad_base_deployment,
create_timeout_deployment,
finish_reason_of,
reasoning_tokens_of,
)
pytestmark = pytest.mark.e2e
@ -34,14 +37,17 @@ def _assert_served_by_fallback(resp: StreamingResponse) -> None:
assert resp.status_code == 200, f"expected 200 after fallback, got {resp.status_code}: {resp.body[:300]}"
content = content_of(resp)
finish_reason = finish_reason_of(resp)
completion_tokens = completion_tokens_of(resp) or 0
reasoning_tokens = reasoning_tokens_of(resp) or 0
assert isinstance(content, str), (
f"the gpt-5.5 fallback should have returned a completion body, got content {content!r} "
f"(body={resp.body[:300]})"
)
assert content or finish_reason == "length", (
f"the gpt-5.5 fallback returned empty content with finish_reason={finish_reason!r}; only "
f"finish_reason='length' may be empty, since gpt-5.5 can spend the whole max_tokens "
f"budget on reasoning (body={resp.body[:300]})"
assert content or (finish_reason == "length" and completion_tokens > 0), (
f"the gpt-5.5 fallback returned empty content with finish_reason={finish_reason!r}, "
f"completion_tokens={completion_tokens}, reasoning_tokens={reasoning_tokens}; empty "
f"content is only acceptable when the budget was spent on non-visible reasoning "
f"(body={resp.body[:300]})"
)
attempted = resp.headers.get("x-litellm-attempted-fallbacks")
assert attempted is not None, "response is missing the x-litellm-attempted-fallbacks header"