mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
[Fix] Responses bridge: wrap chatcmpl-* message item IDs with msg_ envelope
When the Responses API bridge fronts a chat-completions provider, message
output items inherit the upstream ``chatcmpl-*`` id directly. Clients that
follow the OpenAI Responses spec expect typed prefixes (``msg_*``, ``rs_*``)
on output item ids and break when they receive a raw ``chatcmpl-*`` id:
- Vercel AI SDK 6.x fails with "text part {id} not found" on multi-step
tool calls (issue #26529)
- Streaming bridge surfaces unregistered ``chatcmpl-`` ids in text-delta
events (issue #27671)
- Bridge responses can replay ``chatcmpl-*`` ids back into OpenAI on
cross-provider handoffs (issue #27333)
This adds an envelope codec on ``ResponsesAPIRequestUtils`` that wraps raw
upstream ids as ``msg_<base64-payload>`` using the same payload format as
the existing ``_build_responses_api_response_id`` envelope:
- ``_encode_item_envelope(raw_response_id, prefix, custom_llm_provider,
model_id)`` - reuses ``_build_responses_api_response_id`` and swaps
``resp_`` for the requested item prefix (``msg``/``rs``).
- ``_decode_item_envelope(item_id)`` - returns the ``resp_<env>`` form so
existing ``_decode_responses_api_response_id`` machinery handles the
inner payload validation.
- ``_envelope_encode_output_item_ids`` walks the response output and
rewrites only message items whose id is raw (i.e. not already prefixed
with ``msg_``, ``rs_``, or ``encitem_``). Function-call items and
already-prefixed items are left untouched.
Hooked into ``_update_responses_api_response_id_with_model_id`` so the same
post-processor that wraps ``response.id`` now also wraps message item ids,
giving downstream clients consistent typed envelopes across both fields.
- 12 unit tests covering round-trip encode/decode, malformed input
handling, selective rewriting (skip already-prefixed, skip
function_call items), empty output handling, and the full integration
via ``_update_responses_api_response_id_with_model_id``.
This commit is contained in:
parent
10a48f7655
commit
5fb4ff4a22
2 changed files with 347 additions and 0 deletions
|
|
@ -218,6 +218,15 @@ class ResponsesAPIRequestUtils:
|
|||
else:
|
||||
responses_api_response.id = updated_id
|
||||
|
||||
responses_api_response = (
|
||||
ResponsesAPIRequestUtils._envelope_encode_output_item_ids(
|
||||
response=responses_api_response,
|
||||
raw_response_id=response_id,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
model_id=model_id,
|
||||
)
|
||||
)
|
||||
|
||||
if litellm_metadata.get("encrypted_content_affinity_enabled"):
|
||||
responses_api_response = (
|
||||
ResponsesAPIRequestUtils._update_encrypted_content_item_ids_in_response(
|
||||
|
|
@ -228,6 +237,95 @@ class ResponsesAPIRequestUtils:
|
|||
|
||||
return responses_api_response
|
||||
|
||||
@staticmethod
|
||||
def _encode_item_envelope(
|
||||
raw_response_id: str,
|
||||
*,
|
||||
prefix: str,
|
||||
custom_llm_provider: Optional[str],
|
||||
model_id: Optional[str],
|
||||
) -> str:
|
||||
"""Wrap a raw upstream id (e.g. ``chatcmpl-*``) as ``rs_<env>`` or
|
||||
``msg_<env>`` so the value carries the same response_id payload as
|
||||
``response.id`` but with an item-type prefix that clients recognize.
|
||||
|
||||
The envelope reuses
|
||||
:meth:`_build_responses_api_response_id` and swaps the leading
|
||||
``resp_`` for the requested item prefix, so the result round-trips
|
||||
through :meth:`_decode_item_envelope` plus
|
||||
:meth:`_decode_responses_api_response_id` back to ``raw_response_id``.
|
||||
"""
|
||||
resp_form = ResponsesAPIRequestUtils._build_responses_api_response_id(
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
model_id=model_id,
|
||||
response_id=raw_response_id,
|
||||
)
|
||||
return f"{prefix}_" + resp_form[len("resp_") :]
|
||||
|
||||
@staticmethod
|
||||
def _decode_item_envelope(item_id: str) -> Optional[str]:
|
||||
"""Decode ``rs_<env>`` / ``msg_<env>`` back to the ``resp_<env>`` form
|
||||
that :meth:`_decode_responses_api_response_id` understands.
|
||||
|
||||
Returns ``None`` on missing prefix or empty input. All other validation
|
||||
is delegated to the existing response-id decoder.
|
||||
"""
|
||||
if not item_id:
|
||||
return None
|
||||
for prefix in ("rs_", "msg_"):
|
||||
if item_id.startswith(prefix):
|
||||
return "resp_" + item_id[len(prefix) :]
|
||||
return None
|
||||
|
||||
@staticmethod
|
||||
def _envelope_encode_output_item_ids(
|
||||
response: Union["ResponsesAPIResponse", Dict[str, Any]],
|
||||
raw_response_id: str,
|
||||
custom_llm_provider: Optional[str],
|
||||
model_id: Optional[str],
|
||||
) -> Union["ResponsesAPIResponse", Dict[str, Any]]:
|
||||
"""Rewrite output item IDs that leak raw upstream forms (e.g.
|
||||
``chatcmpl-*``) as ``msg_<env>`` so downstream clients see a stable,
|
||||
item-typed identifier instead of a chat-completions artifact.
|
||||
|
||||
Only items whose ``id`` is missing or does not already start with the
|
||||
expected typed prefix (``msg_``, ``rs_``, ``encitem_``) are rewritten.
|
||||
Function-call items keep their ``call_id``-based id (untouched).
|
||||
"""
|
||||
if isinstance(response, dict):
|
||||
output = response.get("output")
|
||||
else:
|
||||
output = getattr(response, "output", None)
|
||||
if not output:
|
||||
return response
|
||||
|
||||
for item in output:
|
||||
if isinstance(item, dict):
|
||||
item_type = item.get("type")
|
||||
current_id = item.get("id")
|
||||
else:
|
||||
item_type = getattr(item, "type", None)
|
||||
current_id = getattr(item, "id", None)
|
||||
if item_type != "message":
|
||||
continue
|
||||
if isinstance(current_id, str) and (
|
||||
current_id.startswith("msg_")
|
||||
or current_id.startswith("rs_")
|
||||
or current_id.startswith("encitem_")
|
||||
):
|
||||
continue
|
||||
new_id = ResponsesAPIRequestUtils._encode_item_envelope(
|
||||
raw_response_id,
|
||||
prefix="msg",
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
model_id=model_id,
|
||||
)
|
||||
if isinstance(item, dict):
|
||||
item["id"] = new_id
|
||||
else:
|
||||
item.id = new_id
|
||||
return response
|
||||
|
||||
@staticmethod
|
||||
def _build_encrypted_item_id(model_id: str, item_id: str) -> str:
|
||||
"""Encode model_id into an output item ID for encrypted-content items.
|
||||
|
|
|
|||
249
tests/test_litellm/responses/test_item_envelope_encoding.py
Normal file
249
tests/test_litellm/responses/test_item_envelope_encoding.py
Normal file
|
|
@ -0,0 +1,249 @@
|
|||
"""Tests for envelope-encoded output item IDs in Responses API responses.
|
||||
|
||||
When the Responses API bridge fronts a chat-completions provider, message
|
||||
output items historically inherit the upstream ``chatcmpl-*`` id from the
|
||||
chat-completion response. Clients that follow the OpenAI Responses spec
|
||||
expect typed prefixes (``msg_*``, ``rs_*``) and break when they receive an
|
||||
``chatcmpl-*`` id (e.g. Vercel AI SDK 6.x "text part {id} not found").
|
||||
|
||||
These tests cover the envelope codec helpers and the integration into
|
||||
``_update_responses_api_response_id_with_model_id``.
|
||||
"""
|
||||
|
||||
from litellm.responses.utils import ResponsesAPIRequestUtils
|
||||
from litellm.types.llms.openai import ResponsesAPIResponse
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Envelope codec (pure)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_encode_decode_round_trip_chatcmpl_id():
|
||||
raw = "chatcmpl-abc123"
|
||||
envelope = ResponsesAPIRequestUtils._encode_item_envelope(
|
||||
raw,
|
||||
prefix="msg",
|
||||
custom_llm_provider="hosted_vllm",
|
||||
model_id="m-1",
|
||||
)
|
||||
assert envelope.startswith("msg_")
|
||||
decoded_resp_form = ResponsesAPIRequestUtils._decode_item_envelope(envelope)
|
||||
assert decoded_resp_form is not None
|
||||
assert decoded_resp_form.startswith("resp_")
|
||||
decoded = ResponsesAPIRequestUtils._decode_responses_api_response_id(
|
||||
decoded_resp_form
|
||||
)
|
||||
assert decoded["response_id"] == raw
|
||||
assert decoded["custom_llm_provider"] == "hosted_vllm"
|
||||
assert decoded["model_id"] == "m-1"
|
||||
|
||||
|
||||
def test_encode_decode_round_trip_reasoning_prefix():
|
||||
raw = "chatcmpl-xyz789"
|
||||
envelope = ResponsesAPIRequestUtils._encode_item_envelope(
|
||||
raw,
|
||||
prefix="rs",
|
||||
custom_llm_provider=None,
|
||||
model_id=None,
|
||||
)
|
||||
assert envelope.startswith("rs_")
|
||||
decoded_resp_form = ResponsesAPIRequestUtils._decode_item_envelope(envelope)
|
||||
assert decoded_resp_form is not None
|
||||
decoded = ResponsesAPIRequestUtils._decode_responses_api_response_id(
|
||||
decoded_resp_form
|
||||
)
|
||||
assert decoded["response_id"] == raw
|
||||
|
||||
|
||||
def test_decode_item_envelope_returns_none_for_missing_prefix():
|
||||
assert ResponsesAPIRequestUtils._decode_item_envelope("chatcmpl-abc") is None
|
||||
assert ResponsesAPIRequestUtils._decode_item_envelope("encitem_abc") is None
|
||||
assert ResponsesAPIRequestUtils._decode_item_envelope("resp_abc") is None
|
||||
|
||||
|
||||
def test_decode_item_envelope_returns_none_for_empty_input():
|
||||
assert ResponsesAPIRequestUtils._decode_item_envelope("") is None
|
||||
|
||||
|
||||
def test_decode_item_envelope_empty_payload_is_resp_passthrough():
|
||||
"""A ``msg_`` prefix with no payload decodes to ``resp_``, which the
|
||||
response-id decoder treats as raw passthrough rather than crashing.
|
||||
"""
|
||||
result = ResponsesAPIRequestUtils._decode_item_envelope("msg_")
|
||||
assert result == "resp_"
|
||||
decoded = ResponsesAPIRequestUtils._decode_responses_api_response_id(result)
|
||||
assert decoded["custom_llm_provider"] is None
|
||||
assert decoded["model_id"] is None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# _envelope_encode_output_item_ids
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _resp_with_message_id(message_id: str) -> ResponsesAPIResponse:
|
||||
return ResponsesAPIResponse(
|
||||
id="chatcmpl-abc",
|
||||
object="response",
|
||||
created_at=0,
|
||||
model="hosted_vllm/test-model",
|
||||
output=[{"type": "message", "id": message_id}],
|
||||
parallel_tool_calls=False,
|
||||
temperature=0,
|
||||
tool_choice="auto",
|
||||
tools=[],
|
||||
top_p=None,
|
||||
max_output_tokens=None,
|
||||
previous_response_id=None,
|
||||
reasoning=None,
|
||||
status="completed",
|
||||
text={},
|
||||
truncation=None,
|
||||
usage=None,
|
||||
user=None,
|
||||
)
|
||||
|
||||
|
||||
def test_envelope_encode_rewrites_chatcmpl_message_id():
|
||||
response = _resp_with_message_id("chatcmpl-abc")
|
||||
encoded = ResponsesAPIRequestUtils._envelope_encode_output_item_ids(
|
||||
response=response,
|
||||
raw_response_id="chatcmpl-abc",
|
||||
custom_llm_provider="hosted_vllm",
|
||||
model_id=None,
|
||||
)
|
||||
new_id = encoded.output[0]["id"]
|
||||
assert new_id.startswith("msg_")
|
||||
# Verify round-trip
|
||||
resp_form = ResponsesAPIRequestUtils._decode_item_envelope(new_id)
|
||||
decoded = ResponsesAPIRequestUtils._decode_responses_api_response_id(resp_form)
|
||||
assert decoded["response_id"] == "chatcmpl-abc"
|
||||
|
||||
|
||||
def test_envelope_encode_skips_already_prefixed_message_id():
|
||||
response = _resp_with_message_id("msg_existing")
|
||||
encoded = ResponsesAPIRequestUtils._envelope_encode_output_item_ids(
|
||||
response=response,
|
||||
raw_response_id="chatcmpl-abc",
|
||||
custom_llm_provider=None,
|
||||
model_id=None,
|
||||
)
|
||||
assert encoded.output[0]["id"] == "msg_existing"
|
||||
|
||||
|
||||
def test_envelope_encode_skips_rs_prefixed_message_id():
|
||||
response = _resp_with_message_id("rs_hash123")
|
||||
encoded = ResponsesAPIRequestUtils._envelope_encode_output_item_ids(
|
||||
response=response,
|
||||
raw_response_id="chatcmpl-abc",
|
||||
custom_llm_provider=None,
|
||||
model_id=None,
|
||||
)
|
||||
assert encoded.output[0]["id"] == "rs_hash123"
|
||||
|
||||
|
||||
def test_envelope_encode_skips_encitem_prefixed_message_id():
|
||||
response = _resp_with_message_id("encitem_abc123")
|
||||
encoded = ResponsesAPIRequestUtils._envelope_encode_output_item_ids(
|
||||
response=response,
|
||||
raw_response_id="chatcmpl-abc",
|
||||
custom_llm_provider=None,
|
||||
model_id=None,
|
||||
)
|
||||
assert encoded.output[0]["id"] == "encitem_abc123"
|
||||
|
||||
|
||||
def test_envelope_encode_leaves_function_call_items_untouched():
|
||||
"""function_call items keep their call_id-based id."""
|
||||
response = ResponsesAPIResponse(
|
||||
id="chatcmpl-abc",
|
||||
object="response",
|
||||
created_at=0,
|
||||
model="hosted_vllm/test-model",
|
||||
output=[
|
||||
{"type": "function_call", "id": "chatcmpl-fc", "call_id": "call_xyz"},
|
||||
{"type": "message", "id": "chatcmpl-abc"},
|
||||
],
|
||||
parallel_tool_calls=False,
|
||||
temperature=0,
|
||||
tool_choice="auto",
|
||||
tools=[],
|
||||
top_p=None,
|
||||
max_output_tokens=None,
|
||||
previous_response_id=None,
|
||||
reasoning=None,
|
||||
status="completed",
|
||||
text={},
|
||||
truncation=None,
|
||||
usage=None,
|
||||
user=None,
|
||||
)
|
||||
encoded = ResponsesAPIRequestUtils._envelope_encode_output_item_ids(
|
||||
response=response,
|
||||
raw_response_id="chatcmpl-abc",
|
||||
custom_llm_provider=None,
|
||||
model_id=None,
|
||||
)
|
||||
assert encoded.output[0]["id"] == "chatcmpl-fc"
|
||||
assert encoded.output[1]["id"].startswith("msg_")
|
||||
|
||||
|
||||
def test_envelope_encode_handles_empty_output():
|
||||
response = ResponsesAPIResponse(
|
||||
id="chatcmpl-abc",
|
||||
object="response",
|
||||
created_at=0,
|
||||
model="hosted_vllm/test-model",
|
||||
output=[],
|
||||
parallel_tool_calls=False,
|
||||
temperature=0,
|
||||
tool_choice="auto",
|
||||
tools=[],
|
||||
top_p=None,
|
||||
max_output_tokens=None,
|
||||
previous_response_id=None,
|
||||
reasoning=None,
|
||||
status="completed",
|
||||
text={},
|
||||
truncation=None,
|
||||
usage=None,
|
||||
user=None,
|
||||
)
|
||||
encoded = ResponsesAPIRequestUtils._envelope_encode_output_item_ids(
|
||||
response=response,
|
||||
raw_response_id="chatcmpl-abc",
|
||||
custom_llm_provider=None,
|
||||
model_id=None,
|
||||
)
|
||||
assert encoded.output == []
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Integration: _update_responses_api_response_id_with_model_id
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_update_responses_api_response_id_rewrites_message_item_id():
|
||||
"""The full update_responses_api_response_id flow wraps both the top-level
|
||||
response.id AND each message item id whose value is a raw chatcmpl-* form.
|
||||
"""
|
||||
response = _resp_with_message_id("chatcmpl-abc")
|
||||
updated = ResponsesAPIRequestUtils._update_responses_api_response_id_with_model_id(
|
||||
responses_api_response=response,
|
||||
custom_llm_provider="hosted_vllm",
|
||||
litellm_metadata={"model_info": {"id": "deploy-1"}},
|
||||
)
|
||||
assert updated.id.startswith("resp_")
|
||||
assert updated.output[0]["id"].startswith("msg_")
|
||||
# The message id and the response id encode the SAME response_id payload
|
||||
msg_resp_form = ResponsesAPIRequestUtils._decode_item_envelope(
|
||||
updated.output[0]["id"]
|
||||
)
|
||||
msg_decoded = ResponsesAPIRequestUtils._decode_responses_api_response_id(
|
||||
msg_resp_form
|
||||
)
|
||||
resp_decoded = ResponsesAPIRequestUtils._decode_responses_api_response_id(
|
||||
updated.id
|
||||
)
|
||||
assert msg_decoded["response_id"] == resp_decoded["response_id"] == "chatcmpl-abc"
|
||||
Loading…
Add table
Reference in a new issue