fix(types): call model_rebuild() on openai SDK streaming types to fix MockValSer crash

openai._models.BaseModel sets defer_build=True, so SDK subclasses start with MockValSer
as their pydantic serializer. model_validate() — used by the SDK for all response parsing
— does not trigger the deferred build (only __init__ does). So streaming objects like
ChoiceLogprobs and ChoiceDeltaToolCall retain MockValSer after parsing.

streaming_handler.py stores these raw SDK objects as extra field values in
ModelResponseStream. When proxy_server._serialize_streaming_chunk() calls model_dump_json(),
pydantic's Rust core looks up type(value).__pydantic_serializer__ at the C level, finds
MockValSer, and raises TypeError.

model_rebuild() at module load time forces the deferred build to complete, installing a
real SchemaSerializer before any request is processed.

Fixes: #24357, #18801, #30617
This commit is contained in:
Lize Cai 2026-06-17 14:18:36 +08:00
parent a9e651d994
commit 7f373bf32c
2 changed files with 124 additions and 0 deletions

View file

@ -3791,3 +3791,53 @@ class GenericGuardrailAPIInputs(TypedDict, total=False):
AllMessageValues
] # structured messages sent to the LLM - indicates if text is from system or user
model: Optional[str] # the model being used for the LLM call
# openai._models.BaseModel sets defer_build=True, so subclasses start with MockValSer
# as their __pydantic_serializer__. model_validate() (used by the SDK for all response
# parsing) does NOT trigger the deferred build — only __init__ does. So every SDK object
# produced by parsing a streaming chunk has MockValSer at the class level.
#
# streaming_handler.py stores these raw SDK objects as extra='allow' field values in
# litellm's ModelResponseStream. When proxy_server calls model_dump_json(), the Rust
# SchemaSerializer looks up type(value).__pydantic_serializer__ at the C level (bypassing
# Python's __getattr__ retry), finds MockValSer, and raises:
# TypeError: 'MockValSer' object cannot be converted to 'SchemaSerializer'
#
# The crash is intermittent because model_dump() (Python path) accidentally self-heals
# via MockValSer.__getattr__ -> attempt_rebuild(). If model_dump() runs first (usage
# chunk), the SDK class is rebuilt as a side effect and model_dump_json() works. If
# model_dump_json() runs first, it crashes.
#
# Remove these calls only when streaming_handler stops storing raw SDK objects in litellm
# models (i.e., always converts to dicts or litellm's own types before assignment).
from openai.types.chat import ChatCompletionChunk as _ChatCompletionChunk
from openai.types.chat.chat_completion_chunk import (
Choice as _OAIChoice,
ChoiceDelta as _OAIChoiceDelta,
ChoiceLogprobs as _OAIChoiceLogprobs,
ChoiceDeltaToolCall as _OAIChoiceDeltaToolCall,
ChoiceDeltaToolCallFunction as _OAIChoiceDeltaToolCallFunction,
)
from openai.types.chat.chat_completion_token_logprob import (
ChatCompletionTokenLogprob as _OAIChatCompletionTokenLogprob,
TopLogprob as _OAITopLogprob,
)
def _rebuild_sdk_streaming_types() -> None:
for cls in (
_ChatCompletionChunk,
_OAIChoice,
_OAIChoiceDelta,
_OAIChoiceLogprobs,
_OAIChoiceDeltaToolCall,
_OAIChoiceDeltaToolCallFunction,
_OAIChatCompletionTokenLogprob,
_OAITopLogprob,
):
cls.model_rebuild()
_rebuild_sdk_streaming_types()
del _rebuild_sdk_streaming_types

View file

@ -344,6 +344,80 @@ def test_parallel_request_limiter_internal_fields_in_all_litellm_params():
)
def test_sdk_streaming_types_model_dump_json_after_model_validate():
"""
Regression test for https://github.com/BerriAI/litellm/issues/18801 and #24357.
SDK objects created via model_validate() carry MockValSer as their pydantic serializer.
When stored as extra fields in ModelResponseStream and serialized via model_dump_json(),
pydantic's Rust core hits MockValSer at the C level and raises TypeError.
"""
from openai.types.chat.chat_completion_chunk import (
ChoiceDeltaToolCall as _OAIChoiceDeltaToolCall,
ChoiceLogprobs as _OAIChoiceLogprobs,
)
from openai.types.chat.chat_completion_token_logprob import (
ChatCompletionTokenLogprob as _OAIChatCompletionTokenLogprob,
)
from litellm.types.utils import Delta, ModelResponseStream, StreamingChoices
logprobs_obj = _OAIChoiceLogprobs.model_validate(
{
"content": [
_OAIChatCompletionTokenLogprob.model_validate(
{"token": "hello", "bytes": [104], "logprob": -0.5, "top_logprobs": []}
)
]
}
)
tool_call_obj = _OAIChoiceDeltaToolCall.model_validate(
{"index": 0, "id": "call_abc", "type": "function", "function": {"name": "my_tool", "arguments": '{"x": 1}'}},
)
model_response = ModelResponseStream(
id="chatcmpl-test",
choices=[StreamingChoices(finish_reason=None, index=0, delta=Delta(role="assistant"), logprobs=logprobs_obj)],
model="gpt-4o",
)
model_response.choices[0].delta.tool_calls = [tool_call_obj]
result = model_response.model_dump_json(exclude_none=True, exclude_unset=True)
assert "my_tool" in result
assert "hello" in result
def test_sdk_streaming_empty_logprobs_model_dump_json():
"""
Regression test for https://github.com/BerriAI/litellm/issues/24357.
The final stop chunk often carries ChoiceLogprobs(content=None, refusal=None).
This empty logprobs object is still a raw SDK type created via model_validate()
and retains MockValSer, causing model_dump_json() to crash the same way.
"""
from openai.types.chat.chat_completion_chunk import ChoiceLogprobs as _OAIChoiceLogprobs
from litellm.types.utils import Delta, ModelResponseStream, StreamingChoices
empty_logprobs = _OAIChoiceLogprobs.model_validate({"content": None, "refusal": None})
model_response = ModelResponseStream(
id="chatcmpl-test",
choices=[
StreamingChoices(
finish_reason="stop",
index=0,
delta=Delta(role="assistant", content=None),
logprobs=empty_logprobs,
)
],
model="gpt-4o",
)
result = model_response.model_dump_json(exclude_none=True, exclude_unset=True)
assert "stop" in result
def test_delta_maps_reasoning_to_reasoning_content():
"""
Test that Delta maps 'reasoning' field to 'reasoning_content'.