mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
fix(types): call model_rebuild() on openai SDK streaming types to fix MockValSer crash
openai._models.BaseModel sets defer_build=True, so SDK subclasses start with MockValSer as their pydantic serializer. model_validate() — used by the SDK for all response parsing — does not trigger the deferred build (only __init__ does). So streaming objects like ChoiceLogprobs and ChoiceDeltaToolCall retain MockValSer after parsing. streaming_handler.py stores these raw SDK objects as extra field values in ModelResponseStream. When proxy_server._serialize_streaming_chunk() calls model_dump_json(), pydantic's Rust core looks up type(value).__pydantic_serializer__ at the C level, finds MockValSer, and raises TypeError. model_rebuild() at module load time forces the deferred build to complete, installing a real SchemaSerializer before any request is processed. Fixes: #24357, #18801, #30617
This commit is contained in:
parent
a9e651d994
commit
7f373bf32c
2 changed files with 124 additions and 0 deletions
|
|
@ -3791,3 +3791,53 @@ class GenericGuardrailAPIInputs(TypedDict, total=False):
|
|||
AllMessageValues
|
||||
] # structured messages sent to the LLM - indicates if text is from system or user
|
||||
model: Optional[str] # the model being used for the LLM call
|
||||
|
||||
|
||||
# openai._models.BaseModel sets defer_build=True, so subclasses start with MockValSer
|
||||
# as their __pydantic_serializer__. model_validate() (used by the SDK for all response
|
||||
# parsing) does NOT trigger the deferred build — only __init__ does. So every SDK object
|
||||
# produced by parsing a streaming chunk has MockValSer at the class level.
|
||||
#
|
||||
# streaming_handler.py stores these raw SDK objects as extra='allow' field values in
|
||||
# litellm's ModelResponseStream. When proxy_server calls model_dump_json(), the Rust
|
||||
# SchemaSerializer looks up type(value).__pydantic_serializer__ at the C level (bypassing
|
||||
# Python's __getattr__ retry), finds MockValSer, and raises:
|
||||
# TypeError: 'MockValSer' object cannot be converted to 'SchemaSerializer'
|
||||
#
|
||||
# The crash is intermittent because model_dump() (Python path) accidentally self-heals
|
||||
# via MockValSer.__getattr__ -> attempt_rebuild(). If model_dump() runs first (usage
|
||||
# chunk), the SDK class is rebuilt as a side effect and model_dump_json() works. If
|
||||
# model_dump_json() runs first, it crashes.
|
||||
#
|
||||
# Remove these calls only when streaming_handler stops storing raw SDK objects in litellm
|
||||
# models (i.e., always converts to dicts or litellm's own types before assignment).
|
||||
from openai.types.chat import ChatCompletionChunk as _ChatCompletionChunk
|
||||
from openai.types.chat.chat_completion_chunk import (
|
||||
Choice as _OAIChoice,
|
||||
ChoiceDelta as _OAIChoiceDelta,
|
||||
ChoiceLogprobs as _OAIChoiceLogprobs,
|
||||
ChoiceDeltaToolCall as _OAIChoiceDeltaToolCall,
|
||||
ChoiceDeltaToolCallFunction as _OAIChoiceDeltaToolCallFunction,
|
||||
)
|
||||
from openai.types.chat.chat_completion_token_logprob import (
|
||||
ChatCompletionTokenLogprob as _OAIChatCompletionTokenLogprob,
|
||||
TopLogprob as _OAITopLogprob,
|
||||
)
|
||||
|
||||
|
||||
def _rebuild_sdk_streaming_types() -> None:
|
||||
for cls in (
|
||||
_ChatCompletionChunk,
|
||||
_OAIChoice,
|
||||
_OAIChoiceDelta,
|
||||
_OAIChoiceLogprobs,
|
||||
_OAIChoiceDeltaToolCall,
|
||||
_OAIChoiceDeltaToolCallFunction,
|
||||
_OAIChatCompletionTokenLogprob,
|
||||
_OAITopLogprob,
|
||||
):
|
||||
cls.model_rebuild()
|
||||
|
||||
|
||||
_rebuild_sdk_streaming_types()
|
||||
del _rebuild_sdk_streaming_types
|
||||
|
|
|
|||
|
|
@ -344,6 +344,80 @@ def test_parallel_request_limiter_internal_fields_in_all_litellm_params():
|
|||
)
|
||||
|
||||
|
||||
def test_sdk_streaming_types_model_dump_json_after_model_validate():
|
||||
"""
|
||||
Regression test for https://github.com/BerriAI/litellm/issues/18801 and #24357.
|
||||
|
||||
SDK objects created via model_validate() carry MockValSer as their pydantic serializer.
|
||||
When stored as extra fields in ModelResponseStream and serialized via model_dump_json(),
|
||||
pydantic's Rust core hits MockValSer at the C level and raises TypeError.
|
||||
"""
|
||||
from openai.types.chat.chat_completion_chunk import (
|
||||
ChoiceDeltaToolCall as _OAIChoiceDeltaToolCall,
|
||||
ChoiceLogprobs as _OAIChoiceLogprobs,
|
||||
)
|
||||
from openai.types.chat.chat_completion_token_logprob import (
|
||||
ChatCompletionTokenLogprob as _OAIChatCompletionTokenLogprob,
|
||||
)
|
||||
|
||||
from litellm.types.utils import Delta, ModelResponseStream, StreamingChoices
|
||||
|
||||
logprobs_obj = _OAIChoiceLogprobs.model_validate(
|
||||
{
|
||||
"content": [
|
||||
_OAIChatCompletionTokenLogprob.model_validate(
|
||||
{"token": "hello", "bytes": [104], "logprob": -0.5, "top_logprobs": []}
|
||||
)
|
||||
]
|
||||
}
|
||||
)
|
||||
tool_call_obj = _OAIChoiceDeltaToolCall.model_validate(
|
||||
{"index": 0, "id": "call_abc", "type": "function", "function": {"name": "my_tool", "arguments": '{"x": 1}'}},
|
||||
)
|
||||
|
||||
model_response = ModelResponseStream(
|
||||
id="chatcmpl-test",
|
||||
choices=[StreamingChoices(finish_reason=None, index=0, delta=Delta(role="assistant"), logprobs=logprobs_obj)],
|
||||
model="gpt-4o",
|
||||
)
|
||||
model_response.choices[0].delta.tool_calls = [tool_call_obj]
|
||||
|
||||
result = model_response.model_dump_json(exclude_none=True, exclude_unset=True)
|
||||
assert "my_tool" in result
|
||||
assert "hello" in result
|
||||
|
||||
|
||||
def test_sdk_streaming_empty_logprobs_model_dump_json():
|
||||
"""
|
||||
Regression test for https://github.com/BerriAI/litellm/issues/24357.
|
||||
|
||||
The final stop chunk often carries ChoiceLogprobs(content=None, refusal=None).
|
||||
This empty logprobs object is still a raw SDK type created via model_validate()
|
||||
and retains MockValSer, causing model_dump_json() to crash the same way.
|
||||
"""
|
||||
from openai.types.chat.chat_completion_chunk import ChoiceLogprobs as _OAIChoiceLogprobs
|
||||
|
||||
from litellm.types.utils import Delta, ModelResponseStream, StreamingChoices
|
||||
|
||||
empty_logprobs = _OAIChoiceLogprobs.model_validate({"content": None, "refusal": None})
|
||||
|
||||
model_response = ModelResponseStream(
|
||||
id="chatcmpl-test",
|
||||
choices=[
|
||||
StreamingChoices(
|
||||
finish_reason="stop",
|
||||
index=0,
|
||||
delta=Delta(role="assistant", content=None),
|
||||
logprobs=empty_logprobs,
|
||||
)
|
||||
],
|
||||
model="gpt-4o",
|
||||
)
|
||||
|
||||
result = model_response.model_dump_json(exclude_none=True, exclude_unset=True)
|
||||
assert "stop" in result
|
||||
|
||||
|
||||
def test_delta_maps_reasoning_to_reasoning_content():
|
||||
"""
|
||||
Test that Delta maps 'reasoning' field to 'reasoning_content'.
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue