mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-05 02:41:56 +00:00
fix(responses): handle response.incomplete streaming event in Responses->Chat transform
The responses->chat streaming transformer handled response.completed but had no branch for response.incomplete. When Azure OpenAI (or any Responses-API compatible provider) returned a response.incomplete event (e.g. due to a content filter or max_output_tokens limit), the code fell through to the "Unhandled event" path, logged a debug line, and returned an empty chunk. This caused the terminal metadata (content_filters, incomplete_details) to be silently dropped and the stream ended without a proper finish_reason. Fix: add an explicit handler for response.incomplete that: - Maps incomplete_details.reason to finish_reason (max_output_tokens -> length, content_filter -> content_filter, anything else -> stop) - Forwards content_filters and incomplete_details via provider_specific_fields so downstream custom loggers and guardrail hooks can inspect them - Extracts and transforms usage if present Fixes #27186
This commit is contained in:
parent
850fe595ac
commit
40bd8d5f31
1 changed files with 51 additions and 0 deletions
|
|
@ -1355,6 +1355,57 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator):
|
|||
],
|
||||
usage=usage,
|
||||
)
|
||||
elif event_type == "response.incomplete":
|
||||
# Response ended early (e.g. content_filter or max_output_tokens).
|
||||
# Map incomplete_details.reason to a finish_reason so downstream
|
||||
# callbacks and guardrails receive a terminal chunk instead of an
|
||||
# empty unhandled-event chunk.
|
||||
response_data = parsed_chunk.get("response", {})
|
||||
incomplete_details = (
|
||||
response_data.get("incomplete_details") if response_data else None
|
||||
)
|
||||
reason = (
|
||||
incomplete_details.get("reason") if incomplete_details else None
|
||||
)
|
||||
# Map Responses API reason -> Chat Completions finish_reason
|
||||
finish_reason: str
|
||||
if reason == "max_output_tokens":
|
||||
finish_reason = "length"
|
||||
elif reason == "content_filter":
|
||||
finish_reason = "content_filter"
|
||||
else:
|
||||
finish_reason = "stop"
|
||||
|
||||
# Surface content_filters and incomplete_details via provider_specific_fields
|
||||
# so that custom loggers and guardrail hooks can inspect them.
|
||||
provider_specific: Dict[str, Any] = {}
|
||||
if incomplete_details:
|
||||
provider_specific["incomplete_details"] = incomplete_details
|
||||
content_filters = (
|
||||
response_data.get("content_filters") if response_data else None
|
||||
)
|
||||
if content_filters:
|
||||
provider_specific["content_filters"] = content_filters
|
||||
|
||||
usage = None
|
||||
if response_data and response_data.get("usage"):
|
||||
from litellm.responses.utils import ResponseAPILoggingUtils
|
||||
|
||||
usage = ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(
|
||||
response_data.get("usage")
|
||||
)
|
||||
|
||||
return ModelResponseStream(
|
||||
choices=[
|
||||
StreamingChoices(
|
||||
index=0,
|
||||
delta=Delta(content=""),
|
||||
finish_reason=finish_reason,
|
||||
)
|
||||
],
|
||||
usage=usage,
|
||||
provider_specific_fields=provider_specific if provider_specific else None,
|
||||
)
|
||||
else:
|
||||
pass
|
||||
# For any unhandled event types, create a minimal valid chunk or skip
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue