From d5aa5337fa0ac5abb68bdaef5cfec923e95a4d5c Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Tue, 26 May 2026 22:18:13 +0000 Subject: [PATCH] fix(responses): assign annotation event sequence_number at emit time Annotation events were assigned a sequence_number when queued, but emitted later (after text/reasoning/tool deltas from subsequent chunks). This caused the queued annotation events to have lower sequence numbers than events emitted before them, breaking the monotonic ordering guarantee for strict clients. Defer assignment to emit time so sequence_number reflects actual emission order. Co-authored-by: Yassin Kortam --- .../streaming_iterator.py | 11 ++++++++--- 1 file changed, 8 insertions(+), 3 deletions(-) diff --git a/litellm/responses/litellm_completion_transformation/streaming_iterator.py b/litellm/responses/litellm_completion_transformation/streaming_iterator.py index fb28468bf8c..e73436400b9 100644 --- a/litellm/responses/litellm_completion_transformation/streaming_iterator.py +++ b/litellm/responses/litellm_completion_transformation/streaming_iterator.py @@ -1079,7 +1079,9 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator): if hasattr(annotation, "model_dump") else dict(annotation) ) - self._sequence_number += 1 + # Sequence number is assigned at emit time (see Priority 4 + # below) to preserve monotonic ordering relative to + # higher-priority events from later chunks. event = OutputTextAnnotationAddedEvent( type=ResponsesAPIStreamEvents.OUTPUT_TEXT_ANNOTATION_ADDED, item_id=item_id, @@ -1087,7 +1089,6 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator): content_index=0, annotation_index=idx, annotation=annotation_dict, - sequence_number=self._sequence_number, ) self._pending_annotation_events.append(event) # Priority 1: Handle reasoning content (highest priority) @@ -1134,12 +1135,16 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator): return self._pending_tool_events.pop(0) # Priority 4: If we have pending annotation events, emit the next one - # This happens when the current chunk has no text/reasoning content + # This happens when the current chunk has no text/reasoning content. + # Assign the sequence number here (at emit time) so it stays monotonic + # relative to other events emitted from intervening chunks. if ( hasattr(self, "_pending_annotation_events") and self._pending_annotation_events ): event = self._pending_annotation_events.pop(0) + self._sequence_number += 1 + event.sequence_number = self._sequence_number return event # Priority 5: If we have pending tool events (from earlier chunk), emit the next one