fix(sap): normalize list-shaped reasoning_content into a string

This commit is contained in:
Devin AI 2026-07-25 10:09:16 +00:00
parent b9b27c2beb
commit e0961ec1b5
3 changed files with 186 additions and 19 deletions

View file

@ -2,12 +2,12 @@ from __future__ import annotations
import json
import time
from typing import AsyncIterator, Iterator, Optional
from typing import Any, AsyncIterator, Iterator, Optional
import httpx
from litellm.llms.base_llm.chat.transformation import BaseConfig, BaseLLMException
from litellm.types.llms.openai import OpenAIChatCompletionChunk
from litellm.types.llms.openai import ChatCompletionThinkingBlock, OpenAIChatCompletionChunk
from ...custom_httpx.llm_http_handler import BaseLLMHTTPHandler
@ -31,6 +31,58 @@ def _now_ts() -> int:
return int(time.time())
def _to_thinking_block(block: dict[str, Any] | str) -> ChatCompletionThinkingBlock:
if isinstance(block, str):
return ChatCompletionThinkingBlock(type="thinking", thinking=block, signature=None)
return ChatCompletionThinkingBlock(
type="thinking",
thinking=block.get("content") or block.get("text") or "",
signature=block.get("signature"),
)
def _normalized_message(message: dict[str, Any]) -> dict[str, Any]:
blocks = message.get("reasoning_content")
if not isinstance(blocks, list):
return message
thinking_blocks = [_to_thinking_block(block) for block in blocks if isinstance(block, (dict, str))]
reasoning_text = "".join(block["thinking"] for block in thinking_blocks)
return {
**message,
"reasoning_content": reasoning_text or None,
"thinking_blocks": message.get("thinking_blocks") or thinking_blocks or None,
}
def normalize_reasoning_content(payload: dict[str, Any]) -> dict[str, Any]:
"""
SAP returns `reasoning_content` as a list of blocks for some models (e.g. `gemini-3.5-flash`),
while the OpenAI-compatible schema expects a string; the structured blocks are kept in
`thinking_blocks`
"""
choices = payload.get("choices")
if not isinstance(choices, list):
return payload
return {
**payload,
"choices": [
{
**choice,
**{
key: _normalized_message(choice[key])
for key in ("message", "delta")
if isinstance(choice.get(key), dict)
},
}
if isinstance(choice, dict)
else choice
for choice in choices
],
}
def _is_terminal_chunk(chunk: OpenAIChatCompletionChunk) -> bool:
"""OpenAI-shaped chunk is terminal if any choice has a non-None finish_reason."""
try:
@ -55,20 +107,22 @@ class _StreamParser:
return None
return OpenAIChatCompletionChunk.model_validate(
{
"id": orc.get("id") or evt.get("request_id") or "stream-chunk",
"object": orc.get("object") or "chat.completion.chunk",
"created": orc.get("created") or evt.get("created") or _now_ts(),
"model": orc.get("model") or "unknown",
"choices": [
{
"index": c.get("index", 0),
"delta": c.get("delta") or {},
"finish_reason": c.get("finish_reason"),
}
for c in (orc.get("choices") or [])
],
}
normalize_reasoning_content(
{
"id": orc.get("id") or evt.get("request_id") or "stream-chunk",
"object": orc.get("object") or "chat.completion.chunk",
"created": orc.get("created") or evt.get("created") or _now_ts(),
"model": orc.get("model") or "unknown",
"choices": [
{
"index": c.get("index", 0),
"delta": c.get("delta") or {},
"finish_reason": c.get("finish_reason"),
}
for c in (orc.get("choices") or [])
],
}
)
)
@staticmethod
@ -92,7 +146,7 @@ class _StreamParser:
# ensure it looks like an OpenAI chunk
if "object" not in fr:
fr["object"] = "chat.completion.chunk"
return OpenAIChatCompletionChunk.model_validate(fr)
return OpenAIChatCompletionChunk.model_validate(normalize_reasoning_content(fr))
# Orchestration incremental delta
if "orchestration_result" in event_obj:
@ -100,7 +154,7 @@ class _StreamParser:
# Already an OpenAI-like chunk
if "choices" in event_obj and "object" in event_obj:
return OpenAIChatCompletionChunk.model_validate(event_obj)
return OpenAIChatCompletionChunk.model_validate(normalize_reasoning_content(event_obj))
# Unknown / heartbeat / metrics
return None

View file

@ -46,6 +46,7 @@ from .handler import (
GenAIHubOrchestrationError,
AsyncSAPStreamIterator,
SAPStreamIterator,
normalize_reasoning_content,
)
# Keys routed outside SAP orchestration `model.params` (prompt, stream, fallbacks, etc.)
@ -402,7 +403,7 @@ class GenAIHubOrchestrationConfig(OpenAIGPTConfig):
original_response=raw_response.text,
additional_args={"complete_input_dict": request_data},
)
response = ModelResponse.model_validate(raw_response.json()["final_result"])
response = ModelResponse.model_validate(normalize_reasoning_content(raw_response.json()["final_result"]))
# Strip markdown code blocks if JSON response_format was used with Anthropic models
# SAP GenAI Hub with Anthropic models sometimes wraps JSON in ```json ... ```

View file

@ -639,3 +639,115 @@ class TestSAPTransformationIntegration:
config["config"]["modules"][1]["translation"]["input"]["type"]
== "sap_document_translation"
)
class TestSAPReasoningContentNormalization:
"""SAP returns reasoning_content as a list of blocks for some models (e.g. gemini-3.5-flash)."""
@pytest.fixture
def mock_config(self):
from litellm.llms.sap.chat.transformation import GenAIHubOrchestrationConfig
config = GenAIHubOrchestrationConfig()
config.token_creator = lambda: "Bearer TEST_TOKEN"
config._base_url = "https://api.test-sap.com"
config._resource_group = "test-group"
return config
def _raw_response(self, reasoning_content):
from unittest.mock import MagicMock
raw_response = MagicMock()
raw_response.json.return_value = {
"final_result": {
"id": "test-id",
"model": "gemini-3.5-flash",
"choices": [
{
"index": 0,
"message": {
"role": "assistant",
"content": "Hello!",
"reasoning_content": reasoning_content,
},
"finish_reason": "stop",
}
],
}
}
raw_response.text = "{}"
return raw_response
def _transform(self, mock_config, raw_response):
from unittest.mock import MagicMock
from litellm.types.utils import ModelResponse
return mock_config.transform_response(
model="gemini-3.5-flash",
raw_response=raw_response,
model_response=ModelResponse(id="test", model="test"),
logging_obj=MagicMock(),
request_data={},
messages=[{"role": "user", "content": "Hi!"}],
optional_params={},
litellm_params={},
encoding=None,
)
def test_list_reasoning_blocks_are_joined_into_a_string(self, mock_config):
raw_response = self._raw_response(
[
{"content": "first thought ", "signature": "sig-1"},
{"content": "second thought", "signature": "sig-2"},
]
)
result = self._transform(mock_config, raw_response)
message = result.choices[0].message
assert message.content == "Hello!"
assert message.reasoning_content == "first thought second thought"
assert message.thinking_blocks == [
{"type": "thinking", "thinking": "first thought ", "signature": "sig-1"},
{"type": "thinking", "thinking": "second thought", "signature": "sig-2"},
]
def test_empty_reasoning_blocks_become_none(self, mock_config):
raw_response = self._raw_response([{"content": "", "signature": "sig"}])
result = self._transform(mock_config, raw_response)
assert getattr(result.choices[0].message, "reasoning_content", None) is None
def test_string_reasoning_content_is_preserved(self, mock_config):
raw_response = self._raw_response("already a string")
result = self._transform(mock_config, raw_response)
assert result.choices[0].message.reasoning_content == "already a string"
def test_streaming_chunk_with_list_reasoning_blocks(self):
from litellm.llms.sap.chat.handler import _StreamParser
chunk = _StreamParser.to_openai_chunk(
{
"orchestration_result": {
"id": "chunk-1",
"model": "gemini-3.5-flash",
"choices": [
{
"index": 0,
"delta": {
"role": "assistant",
"content": "",
"reasoning_content": [{"content": "thinking...", "signature": "sig"}],
},
"finish_reason": None,
}
],
}
}
)
assert chunk is not None
assert chunk.choices[0].delta.reasoning_content == "thinking..."