From c790c6218f918070c7df9bc6cd1a7d5731d87984 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Wed, 20 May 2026 18:30:36 +0000 Subject: [PATCH] fix: surface OCR pricing gaps and recover OUTPUT_TEXT_DONE in ChatGPT SSE - cost_calculator.ocr_cost: log a warning when pages_processed is reported but no ocr_cost_per_page is configured, instead of silently billing zero via an implicit '(... or 0.0) * pages_processed' fallback. Behavior is preserved (zero cost) so free-tier / unpriced models still work, but configuration gaps are now visible in logs. - ChatGPTResponsesAPIConfig._extract_completed_response_from_sse: also collect response.output_text.done events into a text-only items map and merge them into the recovered output (OUTPUT_ITEM_DONE wins on duplicate output_index), mirroring the LiteLLMResponses handler. This recovers text content when a provider only emits OUTPUT_TEXT_DONE and the final response.completed event has an empty output list. Co-authored-by: Yassin Kortam --- litellm/cost_calculator.py | 18 +++- .../llms/chatgpt/responses/transformation.py | 85 ++++++++++++++++++- 2 files changed, 100 insertions(+), 3 deletions(-) diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index 1c88ece795e..612375faa70 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -1903,7 +1903,23 @@ def ocr_cost( return 0.0, 0.0 raise ValueError("OCR response pages_processed is None") - total_ocr_processing_cost: float = (ocr_cost_per_page or 0.0) * pages_processed + if ocr_cost_per_page is None: + # No per-page pricing configured. Either the model is on credit-based + # pricing (and credits weren't returned, so the credit branch above did + # not match) or the model has no OCR pricing entry at all. Surface a + # warning so that missing pricing entries are visible rather than + # silently producing zero cost for billable usage. + verbose_logger.warning( + "OCR cost: model=%s custom_llm_provider=%s reported " + "pages_processed=%s but no ocr_cost_per_page is configured; " + "returning 0.0 cost.", + model, + custom_llm_provider, + pages_processed, + ) + return 0.0, 0.0 + + total_ocr_processing_cost: float = ocr_cost_per_page * pages_processed return total_ocr_processing_cost, 0.0 diff --git a/litellm/llms/chatgpt/responses/transformation.py b/litellm/llms/chatgpt/responses/transformation.py index 27c49ea658d..e35af48ef54 100644 --- a/litellm/llms/chatgpt/responses/transformation.py +++ b/litellm/llms/chatgpt/responses/transformation.py @@ -158,6 +158,7 @@ class ChatGPTResponsesAPIConfig(OpenAIResponsesAPIConfig): completed_response = None error_message = None streamed_output_items: Dict[int, dict] = {} + text_only_output_items: Dict[int, dict] = {} for chunk in body_text.splitlines(): parsed_chunk = self._parse_sse_json_chunk(chunk) if parsed_chunk is None: @@ -173,10 +174,24 @@ class ChatGPTResponsesAPIConfig(OpenAIResponsesAPIConfig): ) continue - if event_type == ResponsesAPIStreamEvents.RESPONSE_COMPLETED: - completed_response = self._build_completed_response_from_chunk( + if event_type == ResponsesAPIStreamEvents.OUTPUT_TEXT_DONE: + self._record_output_text_chunk( parsed_chunk=parsed_chunk, streamed_output_items=streamed_output_items, + text_only_output_items=text_only_output_items, + ) + continue + + if event_type == ResponsesAPIStreamEvents.RESPONSE_COMPLETED: + # Real OUTPUT_ITEM_DONE events take precedence at any given + # output_index, but text-only items at indices without a + # matching OUTPUT_ITEM_DONE must still be preserved (e.g. + # providers that emit only OUTPUT_TEXT_DONE for some indices). + merged_items: Dict[int, dict] = {**text_only_output_items} + merged_items.update(streamed_output_items) + completed_response = self._build_completed_response_from_chunk( + parsed_chunk=parsed_chunk, + streamed_output_items=merged_items, ) break @@ -224,6 +239,72 @@ class ChatGPTResponsesAPIConfig(OpenAIResponsesAPIConfig): index = len(streamed_output_items) streamed_output_items[index] = item + def _record_output_text_chunk( + self, + parsed_chunk: Dict[str, Any], + streamed_output_items: Dict[int, dict], + text_only_output_items: Dict[int, dict], + ) -> None: + text = parsed_chunk.get("text") + if not isinstance(text, str): + return + + try: + output_index_raw = parsed_chunk.get("output_index") + if output_index_raw is None: + raise ValueError("missing output_index") + output_index = int(output_index_raw) + except (TypeError, ValueError): + output_index = len(text_only_output_items) + + # If a real OUTPUT_ITEM_DONE already covered this index, prefer it. + if output_index in streamed_output_items: + return + + item = text_only_output_items.get(output_index) + if item is None: + item = { + "type": "message", + "id": parsed_chunk.get("item_id") or f"msg_{output_index}", + "role": "assistant", + "status": "completed", + "content": [], + } + text_only_output_items[output_index] = item + + content = item.setdefault("content", []) + if not isinstance(content, list): + return + + try: + content_index_raw = parsed_chunk.get("content_index") + if content_index_raw is None: + raise ValueError("missing content_index") + content_index = int(content_index_raw) + except (TypeError, ValueError): + content_index = len(content) + + while len(content) <= content_index: + content.append( + { + "type": "output_text", + "text": "", + "annotations": [], + } + ) + + content_item = content[content_index] + if not isinstance(content_item, dict): + content_item = {} + content[content_index] = content_item + + content_item["type"] = "output_text" + content_item["text"] = text + if parsed_chunk.get("annotations") is not None: + content_item["annotations"] = parsed_chunk["annotations"] + else: + content_item.setdefault("annotations", []) + def _build_completed_response_from_chunk( self, parsed_chunk: Dict[str, Any], streamed_output_items: Dict[int, dict] ) -> Optional[ResponsesAPIResponse]: