mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-05 08:07:05 +00:00
test(e2e): require spent completion tokens before accepting empty fallback content
The first pass accepted any empty completion whose finish_reason was "length", which also swallowed a fallback that produced nothing at all. Require the response to have billed completion tokens as well, so empty content is accepted only when the budget was demonstrably spent on non-visible reasoning. Asserts on completion_tokens rather than reasoning_tokens because the latter is provider-optional; with empty content and a refusal of null, consumed completion tokens are reasoning by elimination, since visible text would be content. Both counts are reported in the failure message. Folds the three body accessors onto one _parsed helper instead of re-parsing per call, and adds completion_tokens_of / reasoning_tokens_of alongside.
This commit is contained in:
parent
6e341b79d8
commit
fd4b540a44
2 changed files with 38 additions and 17 deletions
|
|
@ -66,24 +66,39 @@ def chat_override(
|
|||
)
|
||||
|
||||
|
||||
def _parsed(resp: StreamingResponse) -> ChatResponse | None:
|
||||
try:
|
||||
return ChatResponse.model_validate_json(resp.body)
|
||||
except ValidationError:
|
||||
return None
|
||||
|
||||
|
||||
def content_of(resp: StreamingResponse) -> str | None:
|
||||
"""The assistant message content of a successful chat response, or None when the
|
||||
body is not a success shape (an error body, or an elided streamed body)."""
|
||||
try:
|
||||
parsed = ChatResponse.model_validate_json(resp.body)
|
||||
except ValidationError:
|
||||
return None
|
||||
if not parsed.choices:
|
||||
parsed = _parsed(resp)
|
||||
if parsed is None or not parsed.choices:
|
||||
return None
|
||||
message = parsed.choices[0].message
|
||||
return message.content if message is not None else None
|
||||
|
||||
|
||||
def finish_reason_of(resp: StreamingResponse) -> str | None:
|
||||
try:
|
||||
parsed = ChatResponse.model_validate_json(resp.body)
|
||||
except ValidationError:
|
||||
return None
|
||||
if not parsed.choices:
|
||||
parsed = _parsed(resp)
|
||||
if parsed is None or not parsed.choices:
|
||||
return None
|
||||
return parsed.choices[0].finish_reason
|
||||
|
||||
|
||||
def completion_tokens_of(resp: StreamingResponse) -> int | None:
|
||||
parsed = _parsed(resp)
|
||||
if parsed is None or parsed.usage is None:
|
||||
return None
|
||||
return parsed.usage.completion_tokens
|
||||
|
||||
|
||||
def reasoning_tokens_of(resp: StreamingResponse) -> int | None:
|
||||
parsed = _parsed(resp)
|
||||
if parsed is None or parsed.usage is None or parsed.usage.completion_tokens_details is None:
|
||||
return None
|
||||
return parsed.usage.completion_tokens_details.reasoning_tokens
|
||||
|
|
|
|||
|
|
@ -5,9 +5,10 @@ Each test registers a primary deployment that fails (an unreachable base URL, or
|
|||
a 1ms deadline) and calls it with a `router_settings_override` mapping it to the
|
||||
real `gpt-5.5`. The proof the fallback fired is twofold: the response is a
|
||||
completion from `gpt-5.5`, and the proxy reports at least one attempted fallback
|
||||
in the x-litellm-attempted-fallbacks header. Empty content is accepted only for
|
||||
`finish_reason == "length"`, since gpt-5.5 counts reasoning against max_tokens
|
||||
and can consume the whole budget before emitting any text.
|
||||
in the x-litellm-attempted-fallbacks header. Empty content is accepted only when
|
||||
`finish_reason == "length"` and the response billed completion tokens, since
|
||||
gpt-5.5 counts reasoning against max_tokens and can consume the whole budget
|
||||
before emitting any text; a fallback that produced nothing at all still fails.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
|
@ -21,10 +22,12 @@ from lifecycle import ResourceManager
|
|||
from models import RouterSettingsOverride
|
||||
from reliability_support import (
|
||||
chat_override,
|
||||
completion_tokens_of,
|
||||
content_of,
|
||||
create_bad_base_deployment,
|
||||
create_timeout_deployment,
|
||||
finish_reason_of,
|
||||
reasoning_tokens_of,
|
||||
)
|
||||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
|
@ -34,14 +37,17 @@ def _assert_served_by_fallback(resp: StreamingResponse) -> None:
|
|||
assert resp.status_code == 200, f"expected 200 after fallback, got {resp.status_code}: {resp.body[:300]}"
|
||||
content = content_of(resp)
|
||||
finish_reason = finish_reason_of(resp)
|
||||
completion_tokens = completion_tokens_of(resp) or 0
|
||||
reasoning_tokens = reasoning_tokens_of(resp) or 0
|
||||
assert isinstance(content, str), (
|
||||
f"the gpt-5.5 fallback should have returned a completion body, got content {content!r} "
|
||||
f"(body={resp.body[:300]})"
|
||||
)
|
||||
assert content or finish_reason == "length", (
|
||||
f"the gpt-5.5 fallback returned empty content with finish_reason={finish_reason!r}; only "
|
||||
f"finish_reason='length' may be empty, since gpt-5.5 can spend the whole max_tokens "
|
||||
f"budget on reasoning (body={resp.body[:300]})"
|
||||
assert content or (finish_reason == "length" and completion_tokens > 0), (
|
||||
f"the gpt-5.5 fallback returned empty content with finish_reason={finish_reason!r}, "
|
||||
f"completion_tokens={completion_tokens}, reasoning_tokens={reasoning_tokens}; empty "
|
||||
f"content is only acceptable when the budget was spent on non-visible reasoning "
|
||||
f"(body={resp.body[:300]})"
|
||||
)
|
||||
attempted = resp.headers.get("x-litellm-attempted-fallbacks")
|
||||
assert attempted is not None, "response is missing the x-litellm-attempted-fallbacks header"
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue