fix: dedupe SSE chunk parser and warn on Fireworks tool drop

- Centralize SSE 'data:' chunk parsing in litellm.responses.sse_output_recovery
  so the ChatGPT Responses transformer and the Responses->Chat-Completions bridge
  share a single implementation.
- Log a warning when get_supported_openai_params drops 'tools' for a
  fireworks_ai model whose JSON entry sets supports_function_calling=false,
  so users notice the behavioral change instead of silently losing tools.

Co-authored-by: Yassin Kortam <yassin@berri.ai>
This commit is contained in:
Cursor Agent 2026-05-21 00:55:58 +00:00
parent f3740e2f1b
commit 1f0695d49a
No known key found for this signature in database
4 changed files with 50 additions and 45 deletions

View file

@ -26,12 +26,12 @@ from pydantic import BaseModel
import litellm
from litellm import ModelResponse
from litellm._logging import verbose_logger
from litellm.litellm_core_utils.streaming_handler import CustomStreamWrapper
from litellm.llms.base_llm.base_model_iterator import BaseModelResponseIterator
from litellm.llms.base_llm.bridges.completion_transformation import (
CompletionTransformationBridge,
)
from litellm.responses.sse_output_recovery import (
parse_sse_json_chunk,
record_output_item_chunk,
record_output_text_chunk,
)
@ -606,25 +606,6 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
return choices
@classmethod
def _parse_raw_sse_chunk(cls, chunk: str) -> Optional[Dict[str, Any]]:
stripped_chunk = (
CustomStreamWrapper._strip_sse_data_from_chunk(chunk.strip()) or ""
).strip()
if (
not stripped_chunk
or stripped_chunk == "[DONE]"
or stripped_chunk.startswith("event:")
):
return None
try:
parsed_chunk = json.loads(stripped_chunk)
except json.JSONDecodeError:
return None
if not isinstance(parsed_chunk, dict):
return None
return parsed_chunk
@classmethod
def _extract_output_from_completed_event(
cls, parsed_chunk: Dict[str, Any]
@ -648,7 +629,7 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
recovered_text_only_items: Dict[int, Dict[str, Any]] = {}
for chunk in raw_sse.splitlines():
parsed_chunk = cls._parse_raw_sse_chunk(chunk)
parsed_chunk = parse_sse_json_chunk(chunk)
if parsed_chunk is None:
continue

View file

@ -1,7 +1,5 @@
import json
from typing import Any, Dict, Optional
from litellm.constants import STREAM_SSE_DONE_STRING
from litellm.exceptions import AuthenticationError
from litellm.litellm_core_utils.core_helpers import process_response_headers
from litellm.litellm_core_utils.llm_response_utils.convert_dict_to_response import (
@ -10,6 +8,7 @@ from litellm.litellm_core_utils.llm_response_utils.convert_dict_to_response impo
from litellm.llms.openai.common_utils import OpenAIError
from litellm.llms.openai.responses.transformation import OpenAIResponsesAPIConfig
from litellm.responses.sse_output_recovery import (
parse_sse_json_chunk,
record_output_item_chunk,
record_output_text_chunk,
)
@ -19,7 +18,6 @@ from litellm.types.llms.openai import (
)
from litellm.types.router import GenericLiteLLMParams
from litellm.types.utils import LlmProviders
from litellm.utils import CustomStreamWrapper
from ..authenticator import Authenticator
from ..common_utils import (
@ -164,7 +162,7 @@ class ChatGPTResponsesAPIConfig(OpenAIResponsesAPIConfig):
streamed_output_items: Dict[int, dict] = {}
text_only_output_items: Dict[int, dict] = {}
for chunk in body_text.splitlines():
parsed_chunk = self._parse_sse_json_chunk(chunk)
parsed_chunk = parse_sse_json_chunk(chunk)
if parsed_chunk is None:
continue
@ -207,25 +205,6 @@ class ChatGPTResponsesAPIConfig(OpenAIResponsesAPIConfig):
return completed_response, error_message
def _parse_sse_json_chunk(self, chunk: str) -> Optional[Dict[str, Any]]:
# Strip outer whitespace before removing the SSE `data:` prefix.
# `_strip_sse_data_from_chunk` only matches the prefix at position 0,
# so chunks with leading whitespace (e.g. ` data: {...}`) would
# otherwise be returned unchanged and fail JSON parsing silently.
stripped_chunk = CustomStreamWrapper._strip_sse_data_from_chunk(chunk.strip())
if not stripped_chunk:
return None
stripped_chunk = stripped_chunk.strip()
if not stripped_chunk or stripped_chunk == STREAM_SSE_DONE_STRING:
return None
try:
parsed_chunk = json.loads(stripped_chunk)
except json.JSONDecodeError:
return None
if not isinstance(parsed_chunk, dict):
return None
return parsed_chunk
def _build_completed_response_from_chunk(
self, parsed_chunk: Dict[str, Any], streamed_output_items: Dict[int, dict]
) -> Optional[ResponsesAPIResponse]:

View file

@ -4,6 +4,7 @@ from typing import Any, List, Literal, Optional, Tuple, Union, cast
import httpx
import litellm
from litellm._logging import verbose_logger
from litellm._uuid import uuid
from litellm.constants import RESPONSE_FORMAT_TOOL_NAME
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
@ -114,6 +115,18 @@ class FireworksAIConfig(OpenAIGPTConfig):
if supports_function_calling(model=model, custom_llm_provider="fireworks_ai"):
supported_params.append("tools")
supported_params.append("parallel_tool_calls")
else:
# Historically every Fireworks model advertised tool support, so a
# JSON entry that flips `supports_function_calling` to false will
# silently drop `tools` from requests. Surface this so users can
# tell why their tool calls suddenly stop working.
verbose_logger.warning(
"fireworks_ai model %r is marked as not supporting "
"function calling in model_prices_and_context_window.json; "
"`tools` and `parallel_tool_calls` will be dropped from the "
"request.",
model,
)
# Only add tool_choice for models that explicitly support it
if supports_tool_choice(model=model, custom_llm_provider="fireworks_ai"):

View file

@ -7,11 +7,43 @@ bridge). Keep the implementation in a single module so a fix in one
caller automatically applies to all of them.
"""
from typing import Any, Dict
import json
from typing import Any, Dict, Optional
from litellm.constants import STREAM_SSE_DONE_STRING
_MAX_CONTENT_INDEX = 1024
def parse_sse_json_chunk(chunk: str) -> Optional[Dict[str, Any]]:
"""Parse a single raw SSE line into a JSON object dict.
Returns ``None`` for empty lines, ``event:`` lines, ``[DONE]`` markers,
invalid JSON, or non-dict payloads. Centralizes the parsing step that
feeds into the recovery helpers in this module so behavior stays
consistent across all callers.
"""
# Import locally to avoid a circular import with the streaming handler.
from litellm.litellm_core_utils.streaming_handler import CustomStreamWrapper
stripped_chunk = (
CustomStreamWrapper._strip_sse_data_from_chunk(chunk.strip()) or ""
).strip()
if (
not stripped_chunk
or stripped_chunk == STREAM_SSE_DONE_STRING
or stripped_chunk.startswith("event:")
):
return None
try:
parsed_chunk = json.loads(stripped_chunk)
except json.JSONDecodeError:
return None
if not isinstance(parsed_chunk, dict):
return None
return parsed_chunk
def record_output_item_chunk(
parsed_chunk: Dict[str, Any],
output_items: Dict[int, Dict[str, Any]],