From 1f0695d49ac4d1127d8861308d2468a288868799 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Thu, 21 May 2026 00:55:58 +0000 Subject: [PATCH] fix: dedupe SSE chunk parser and warn on Fireworks tool drop - Centralize SSE 'data:' chunk parsing in litellm.responses.sse_output_recovery so the ChatGPT Responses transformer and the Responses->Chat-Completions bridge share a single implementation. - Log a warning when get_supported_openai_params drops 'tools' for a fireworks_ai model whose JSON entry sets supports_function_calling=false, so users notice the behavioral change instead of silently losing tools. Co-authored-by: Yassin Kortam --- .../transformation.py | 23 ++----------- .../llms/chatgpt/responses/transformation.py | 25 ++------------ .../llms/fireworks_ai/chat/transformation.py | 13 +++++++ litellm/responses/sse_output_recovery.py | 34 ++++++++++++++++++- 4 files changed, 50 insertions(+), 45 deletions(-) diff --git a/litellm/completion_extras/litellm_responses_transformation/transformation.py b/litellm/completion_extras/litellm_responses_transformation/transformation.py index b3f00d760b4..51abbbf729b 100644 --- a/litellm/completion_extras/litellm_responses_transformation/transformation.py +++ b/litellm/completion_extras/litellm_responses_transformation/transformation.py @@ -26,12 +26,12 @@ from pydantic import BaseModel import litellm from litellm import ModelResponse from litellm._logging import verbose_logger -from litellm.litellm_core_utils.streaming_handler import CustomStreamWrapper from litellm.llms.base_llm.base_model_iterator import BaseModelResponseIterator from litellm.llms.base_llm.bridges.completion_transformation import ( CompletionTransformationBridge, ) from litellm.responses.sse_output_recovery import ( + parse_sse_json_chunk, record_output_item_chunk, record_output_text_chunk, ) @@ -606,25 +606,6 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): return choices - @classmethod - def _parse_raw_sse_chunk(cls, chunk: str) -> Optional[Dict[str, Any]]: - stripped_chunk = ( - CustomStreamWrapper._strip_sse_data_from_chunk(chunk.strip()) or "" - ).strip() - if ( - not stripped_chunk - or stripped_chunk == "[DONE]" - or stripped_chunk.startswith("event:") - ): - return None - try: - parsed_chunk = json.loads(stripped_chunk) - except json.JSONDecodeError: - return None - if not isinstance(parsed_chunk, dict): - return None - return parsed_chunk - @classmethod def _extract_output_from_completed_event( cls, parsed_chunk: Dict[str, Any] @@ -648,7 +629,7 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): recovered_text_only_items: Dict[int, Dict[str, Any]] = {} for chunk in raw_sse.splitlines(): - parsed_chunk = cls._parse_raw_sse_chunk(chunk) + parsed_chunk = parse_sse_json_chunk(chunk) if parsed_chunk is None: continue diff --git a/litellm/llms/chatgpt/responses/transformation.py b/litellm/llms/chatgpt/responses/transformation.py index e0b5b7ac65e..56b61b66c84 100644 --- a/litellm/llms/chatgpt/responses/transformation.py +++ b/litellm/llms/chatgpt/responses/transformation.py @@ -1,7 +1,5 @@ -import json from typing import Any, Dict, Optional -from litellm.constants import STREAM_SSE_DONE_STRING from litellm.exceptions import AuthenticationError from litellm.litellm_core_utils.core_helpers import process_response_headers from litellm.litellm_core_utils.llm_response_utils.convert_dict_to_response import ( @@ -10,6 +8,7 @@ from litellm.litellm_core_utils.llm_response_utils.convert_dict_to_response impo from litellm.llms.openai.common_utils import OpenAIError from litellm.llms.openai.responses.transformation import OpenAIResponsesAPIConfig from litellm.responses.sse_output_recovery import ( + parse_sse_json_chunk, record_output_item_chunk, record_output_text_chunk, ) @@ -19,7 +18,6 @@ from litellm.types.llms.openai import ( ) from litellm.types.router import GenericLiteLLMParams from litellm.types.utils import LlmProviders -from litellm.utils import CustomStreamWrapper from ..authenticator import Authenticator from ..common_utils import ( @@ -164,7 +162,7 @@ class ChatGPTResponsesAPIConfig(OpenAIResponsesAPIConfig): streamed_output_items: Dict[int, dict] = {} text_only_output_items: Dict[int, dict] = {} for chunk in body_text.splitlines(): - parsed_chunk = self._parse_sse_json_chunk(chunk) + parsed_chunk = parse_sse_json_chunk(chunk) if parsed_chunk is None: continue @@ -207,25 +205,6 @@ class ChatGPTResponsesAPIConfig(OpenAIResponsesAPIConfig): return completed_response, error_message - def _parse_sse_json_chunk(self, chunk: str) -> Optional[Dict[str, Any]]: - # Strip outer whitespace before removing the SSE `data:` prefix. - # `_strip_sse_data_from_chunk` only matches the prefix at position 0, - # so chunks with leading whitespace (e.g. ` data: {...}`) would - # otherwise be returned unchanged and fail JSON parsing silently. - stripped_chunk = CustomStreamWrapper._strip_sse_data_from_chunk(chunk.strip()) - if not stripped_chunk: - return None - stripped_chunk = stripped_chunk.strip() - if not stripped_chunk or stripped_chunk == STREAM_SSE_DONE_STRING: - return None - try: - parsed_chunk = json.loads(stripped_chunk) - except json.JSONDecodeError: - return None - if not isinstance(parsed_chunk, dict): - return None - return parsed_chunk - def _build_completed_response_from_chunk( self, parsed_chunk: Dict[str, Any], streamed_output_items: Dict[int, dict] ) -> Optional[ResponsesAPIResponse]: diff --git a/litellm/llms/fireworks_ai/chat/transformation.py b/litellm/llms/fireworks_ai/chat/transformation.py index 881ee27f07f..4a01a3d3b84 100644 --- a/litellm/llms/fireworks_ai/chat/transformation.py +++ b/litellm/llms/fireworks_ai/chat/transformation.py @@ -4,6 +4,7 @@ from typing import Any, List, Literal, Optional, Tuple, Union, cast import httpx import litellm +from litellm._logging import verbose_logger from litellm._uuid import uuid from litellm.constants import RESPONSE_FORMAT_TOOL_NAME from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj @@ -114,6 +115,18 @@ class FireworksAIConfig(OpenAIGPTConfig): if supports_function_calling(model=model, custom_llm_provider="fireworks_ai"): supported_params.append("tools") supported_params.append("parallel_tool_calls") + else: + # Historically every Fireworks model advertised tool support, so a + # JSON entry that flips `supports_function_calling` to false will + # silently drop `tools` from requests. Surface this so users can + # tell why their tool calls suddenly stop working. + verbose_logger.warning( + "fireworks_ai model %r is marked as not supporting " + "function calling in model_prices_and_context_window.json; " + "`tools` and `parallel_tool_calls` will be dropped from the " + "request.", + model, + ) # Only add tool_choice for models that explicitly support it if supports_tool_choice(model=model, custom_llm_provider="fireworks_ai"): diff --git a/litellm/responses/sse_output_recovery.py b/litellm/responses/sse_output_recovery.py index fb7b8e5d850..5c18770a611 100644 --- a/litellm/responses/sse_output_recovery.py +++ b/litellm/responses/sse_output_recovery.py @@ -7,11 +7,43 @@ bridge). Keep the implementation in a single module so a fix in one caller automatically applies to all of them. """ -from typing import Any, Dict +import json +from typing import Any, Dict, Optional + +from litellm.constants import STREAM_SSE_DONE_STRING _MAX_CONTENT_INDEX = 1024 +def parse_sse_json_chunk(chunk: str) -> Optional[Dict[str, Any]]: + """Parse a single raw SSE line into a JSON object dict. + + Returns ``None`` for empty lines, ``event:`` lines, ``[DONE]`` markers, + invalid JSON, or non-dict payloads. Centralizes the parsing step that + feeds into the recovery helpers in this module so behavior stays + consistent across all callers. + """ + # Import locally to avoid a circular import with the streaming handler. + from litellm.litellm_core_utils.streaming_handler import CustomStreamWrapper + + stripped_chunk = ( + CustomStreamWrapper._strip_sse_data_from_chunk(chunk.strip()) or "" + ).strip() + if ( + not stripped_chunk + or stripped_chunk == STREAM_SSE_DONE_STRING + or stripped_chunk.startswith("event:") + ): + return None + try: + parsed_chunk = json.loads(stripped_chunk) + except json.JSONDecodeError: + return None + if not isinstance(parsed_chunk, dict): + return None + return parsed_chunk + + def record_output_item_chunk( parsed_chunk: Dict[str, Any], output_items: Dict[int, Dict[str, Any]],