mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
(fix): empty response + vllm streaming (#17516)
* Fix empty response + vllm streaming * Add unit test
This commit is contained in:
parent
7106509ef0
commit
6af693de36
2 changed files with 32 additions and 2 deletions
|
|
@ -444,7 +444,7 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM):
|
|||
else:
|
||||
headers = {}
|
||||
response = raw_response.parse()
|
||||
if not hasattr(response, "model_dump"):
|
||||
if not data.get("stream") and not hasattr(response, "model_dump"):
|
||||
raise OpenAIError(
|
||||
status_code=500,
|
||||
message=f"Empty or invalid response from LLM endpoint. Received: {response!r}. Check the reverse proxy or model server configuration.",
|
||||
|
|
@ -482,7 +482,7 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM):
|
|||
else:
|
||||
headers = {}
|
||||
response = raw_response.parse()
|
||||
if not hasattr(response, "model_dump"):
|
||||
if not data.get("stream") and not hasattr(response, "model_dump"):
|
||||
raise OpenAIError(
|
||||
status_code=500,
|
||||
message=f"Empty or invalid response from LLM endpoint. Received: {response!r}. Check the reverse proxy or model server configuration.",
|
||||
|
|
|
|||
|
|
@ -95,3 +95,33 @@ class TestEmptyResponseHandling:
|
|||
|
||||
assert response == mock_response
|
||||
assert headers == {"x-request-id": "123"}
|
||||
|
||||
def test_sync_streaming_response_passes_through_without_model_dump(self):
|
||||
"""
|
||||
Test that streaming responses (which don't have model_dump) pass through
|
||||
correctly without raising an error. This validates the fix for VLLM streaming.
|
||||
"""
|
||||
openai_chat = OpenAIChatCompletion()
|
||||
|
||||
# Create a mock response WITHOUT model_dump (like an AsyncStream/Iterator)
|
||||
mock_stream = MagicMock(spec=[]) # spec=[] means no attributes
|
||||
|
||||
mock_raw_response = MagicMock()
|
||||
mock_raw_response.headers = {"x-request-id": "123"}
|
||||
mock_raw_response.parse.return_value = mock_stream
|
||||
|
||||
mock_client = MagicMock()
|
||||
mock_client.chat.completions.with_raw_response.create.return_value = (
|
||||
mock_raw_response
|
||||
)
|
||||
|
||||
# Key: data has stream=True - this should bypass the model_dump check
|
||||
headers, response = openai_chat.make_sync_openai_chat_completion_request(
|
||||
openai_client=mock_client,
|
||||
data={"messages": [{"role": "user", "content": "test"}], "stream": True},
|
||||
timeout=30,
|
||||
logging_obj=MagicMock(),
|
||||
)
|
||||
|
||||
assert response == mock_stream
|
||||
assert headers == {"x-request-id": "123"}
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue