fix(responses-bridge): keep reasoning text visible to inspection-only callers

Guardrails, token counting and rate limiting share the input transform with
the provider path, so moving reasoning onto reasoning_content hid it from
them. Provider-bound callers opt in with replay_reasoning.
This commit is contained in:
Mateo Wang 2026-08-22 11:39:10 -07:00
parent a7afe986e3
commit ae25da3d54
3 changed files with 79 additions and 4 deletions

View file

@ -113,6 +113,7 @@ class ResponsesSessionHandler:
chat_completion_messages = LiteLLMCompletionResponsesConfig.transform_responses_api_input_to_messages(
input=response_input_param,
responses_api_request=proxy_server_request_dict or {},
replay_reasoning=True,
)
chat_completion_message_history.extend(chat_completion_messages)
@ -125,6 +126,7 @@ class ResponsesSessionHandler:
chat_completion_messages = LiteLLMCompletionResponsesConfig.transform_responses_api_input_to_messages(
input=_messages,
responses_api_request=proxy_server_request_dict or {},
replay_reasoning=True,
)
chat_completion_message_history.extend(chat_completion_messages)

View file

@ -296,6 +296,7 @@ class LiteLLMCompletionResponsesConfig:
"messages": LiteLLMCompletionResponsesConfig.transform_responses_api_input_to_messages(
input=input,
responses_api_request=responses_api_request,
replay_reasoning=True,
),
"model": model,
"tool_choice": LiteLLMCompletionResponsesConfig._transform_tool_choice(
@ -340,6 +341,7 @@ class LiteLLMCompletionResponsesConfig:
def transform_responses_api_input_to_messages(
input: str | ResponseInputParam,
responses_api_request: ResponsesAPIOptionalRequestParams | dict,
replay_reasoning: bool = False,
) -> list[
AllMessageValues
| GenericChatCompletionMessage
@ -349,6 +351,16 @@ class LiteLLMCompletionResponsesConfig:
]:
"""
Transform a Responses API input into a list of messages
``replay_reasoning`` belongs to callers whose messages are about to be
sent to a model: prior-turn ``reasoning`` items are then rebuilt as
assistant ``reasoning_content`` and signed ``thinking_blocks`` so the
provider gets its own chain-of-thought back instead of reading it as
visible text.
Callers that only inspect the messages (token counting, rate limiting,
guardrail scanning) leave it off, because they need every piece of text
in the request to stay readable as message ``content``.
"""
messages: list[
AllMessageValues
@ -367,6 +379,7 @@ class LiteLLMCompletionResponsesConfig:
messages.extend(
LiteLLMCompletionResponsesConfig._transform_response_input_param_to_chat_completion_message(
input=input,
replay_reasoning=replay_reasoning,
)
)
@ -443,11 +456,15 @@ class LiteLLMCompletionResponsesConfig:
@staticmethod
def _transform_response_input_param_to_chat_completion_message(
input: str | ResponseInputParam,
replay_reasoning: bool = False,
) -> list[
AllMessageValues | GenericChatCompletionMessage | ChatCompletionMessageToolCall | ChatCompletionResponseMessage
]:
"""
Transform a ResponseInputParam into a Chat Completion message
See ``transform_responses_api_input_to_messages`` for what
``replay_reasoning`` means.
"""
messages: list[
AllMessageValues
@ -463,7 +480,8 @@ class LiteLLMCompletionResponsesConfig:
for _input in input:
chat_completion_messages = (
LiteLLMCompletionResponsesConfig._transform_responses_api_input_item_to_chat_completion_message(
input_item=_input
input_item=_input,
replay_reasoning=replay_reasoning,
)
)
@ -559,6 +577,8 @@ class LiteLLMCompletionResponsesConfig:
continue
messages.extend(chat_completion_messages)
if not replay_reasoning:
return messages
return LiteLLMCompletionResponsesConfig._merge_reasoning_only_assistant_messages(messages)
@staticmethod
@ -1151,6 +1171,7 @@ class LiteLLMCompletionResponsesConfig:
@staticmethod
def _transform_responses_api_input_item_to_chat_completion_message(
input_item: Any,
replay_reasoning: bool = False,
) -> list[AllMessageValues | GenericChatCompletionMessage | ChatCompletionResponseMessage]:
"""
Transform a Responses API input item into a Chat Completion message
@ -1179,12 +1200,14 @@ class LiteLLMCompletionResponsesConfig:
return LiteLLMCompletionResponsesConfig._transform_responses_api_function_call_to_chat_completion_message(
function_call=input_item
)
elif input_item.get("type") == "reasoning":
elif replay_reasoning and input_item.get("type") == "reasoning":
# A ResponseReasoningItemParam carries the prior-turn chain-of-thought.
# Chat-completions providers (DeepSeek V4, Kimi K2.6, ...) expect this
# to be replayed as `reasoning_content` on an assistant message, not as
# visible `content` (prompt pollution) and not dropped (DeepSeek V4
# rejects multi-turn requests with a missing `reasoning_content`).
# Callers that only inspect the request skip this branch so the
# reasoning text stays visible to them as message content.
reasoning_text = LiteLLMCompletionResponsesConfig._extract_reasoning_text_from_input_item( # rebind-ok: extraction result
input_item
)

View file

@ -23,13 +23,19 @@ from litellm.types.utils import Message
def _transform_item(item):
return LiteLLMCompletionResponsesConfig._transform_responses_api_input_item_to_chat_completion_message(
input_item=item
input_item=item, replay_reasoning=True
)
def _transform_input(input_items):
return LiteLLMCompletionResponsesConfig._transform_response_input_param_to_chat_completion_message(
input=input_items
input=input_items, replay_reasoning=True
)
def _inspect_input(input_items):
return LiteLLMCompletionResponsesConfig.transform_responses_api_input_to_messages(
input=input_items, responses_api_request={}
)
@ -266,6 +272,50 @@ class TestEncryptedReasoningRoundTrip:
assert messages[1]["role"] == "user"
class TestInspectionCallersStillSeeReasoningText:
"""Token counting, rate limiting and guardrails read the request as text.
Moving reasoning onto ``reasoning_content`` is only right for messages on
their way to a provider. A guardrail scanning for sensitive data reads
message ``content``, so the inspection default keeps the text there.
"""
def test_reasoning_text_stays_readable_as_content_by_default(self):
messages = _inspect_input(
[
{"role": "user", "content": "What did we decide?"},
{
"type": "reasoning",
"id": "rs_1",
"content": [{"type": "output_text", "text": "card 4111111111111111"}],
},
]
)
assert len(messages) == 2
blocks = messages[1]["content"]
assert "4111111111111111" in json.dumps(blocks)
assert "reasoning_content" not in messages[1]
def test_reasoning_moves_off_content_only_for_provider_bound_callers(self):
input_items = [
{
"type": "reasoning",
"id": "rs_1",
"content": [{"type": "output_text", "text": "hidden plan"}],
},
{"role": "user", "content": "go on"},
]
provider_bound = LiteLLMCompletionResponsesConfig.transform_responses_api_input_to_messages(
input=input_items, responses_api_request={}, replay_reasoning=True
)
assert provider_bound[0]["content"] is None
assert provider_bound[0]["reasoning_content"] == "hidden plan"
inspected = _inspect_input(input_items)
assert inspected[0]["role"] == "user"
assert "hidden plan" in json.dumps(inspected[0]["content"])
class TestNonReasoningInputItemUnchanged:
"""Non-reasoning items still flow through the existing branches."""