fix(responses): keep the addressed response id off bridged provider requests

The Responses id security hook keeps the id a client addressed under
`_litellm_addressed_response_id` in the request body so internal retries can
re-authorize it. On a model without a native Responses config that body is
bridged into `completion()` kwargs, the key was treated as a provider param,
and providers rejected it, so every follow-up turn carrying
`previous_response_id` returned 400.

Register the key in `all_litellm_params` so it is dropped before any provider
request, and share one constant between the hook and the param list.
This commit is contained in:
mateo-berri 2026-09-17 15:35:11 -07:00
parent 1dd4c13815
commit ca91751d5b
4 changed files with 89 additions and 6 deletions

View file

@ -22,7 +22,7 @@ from litellm.types.llms.openai import (
BaseLiteLLMOpenAIResponseObject,
ResponsesAPIResponse,
)
from litellm.types.utils import CallTypesLiteral, LLMResponseTypes, SpecialEnums
from litellm.types.utils import ADDRESSED_RESPONSE_ID_FIELD, CallTypesLiteral, LLMResponseTypes, SpecialEnums
if TYPE_CHECKING:
from litellm.caching.caching import DualCache
@ -32,7 +32,6 @@ if TYPE_CHECKING:
_RESPONSES_API_PROVIDER_PREFIX: Final = "/openai"
_RESPONSES_API_CREATE_ROUTES: Final = frozenset({"/v1/responses", "/responses"})
_ADDRESSED_RESPONSE_ID_KEY: Final = "_litellm_addressed_response_id"
_UNMANAGED_RESPONSE_ID_DETAIL: Final = (
"Forbidden. This response id was not issued by this proxy, so the proxy cannot tell who owns it. "
"To let keys address responses this proxy did not issue, set "
@ -132,7 +131,7 @@ class ResponsesIDSecurity(CustomLogger):
if call_type not in responses_api_call_types:
return None
addressed_id_field: Final = "previous_response_id" if call_type == "aresponses" else "response_id"
retained_id: Final = data.get(_ADDRESSED_RESPONSE_ID_KEY)
retained_id: Final = data.get(ADDRESSED_RESPONSE_ID_FIELD)
addressed_id: Final = (
retained_id if isinstance(retained_id, str) and retained_id else data.get(addressed_id_field)
)
@ -140,7 +139,7 @@ class ResponsesIDSecurity(CustomLogger):
return data
authorized_id: Final = self._authorize_response_id(addressed_id, user_api_key_dict)
data[addressed_id_field] = authorized_id
data[_ADDRESSED_RESPONSE_ID_KEY] = addressed_id
data[ADDRESSED_RESPONSE_ID_FIELD] = addressed_id
return data
def _authorize_response_id(

View file

@ -3754,6 +3754,8 @@ agentic_loop_internal_litellm_params: Final = [
# the provider.
TRUSTED_CALLBACK_VARS_FIELD: Final = "litellm_trusted_callback_vars"
ADDRESSED_RESPONSE_ID_FIELD: Final = "_litellm_addressed_response_id"
# Bedrock managed-batch deployment config, read from litellm_params by the batch and
# files transformations. Listed for the same reason as the fields above: these sit on
# a deployment that also serves chat, so leaking them into extra_body makes Bedrock
@ -3768,7 +3770,7 @@ bedrock_batch_litellm_params: Final = (
all_litellm_params = (
agentic_loop_internal_litellm_params
+ [TRUSTED_CALLBACK_VARS_FIELD, *bedrock_batch_litellm_params]
+ [TRUSTED_CALLBACK_VARS_FIELD, ADDRESSED_RESPONSE_ID_FIELD, *bedrock_batch_litellm_params]
+ [
"metadata",
"litellm_metadata",

View file

@ -12,14 +12,21 @@ capture the forwarded kwargs; if the flag-setting line is removed the captured
kwargs lack the flag and these tests fail.
"""
import json
from collections.abc import Mapping
from typing import Final
from unittest.mock import patch
import httpx
import pytest
import litellm
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler
from litellm.responses.litellm_completion_transformation.handler import (
LiteLLMCompletionTransformationHandler,
)
from litellm.types.llms.openai import ResponsesAPIResponse
from litellm.types.utils import ADDRESSED_RESPONSE_ID_FIELD
class _StopForwarding(Exception):
@ -170,3 +177,56 @@ async def test_async_fallback_returns_hoisted_nested_custom_tool_call_as_custom_
tool_calls = [(item.type, item.name, item.input) for item in response.output if item.type == "custom_tool_call"]
assert tool_calls == [("custom_tool_call", "exec", "ls")]
class _RecordingAnthropicHandler:
def __init__(self, reply: Mapping[str, object]) -> None:
self.reply: Final = reply
self.request_body: Mapping[str, object] | None = None
def __call__(self, request: httpx.Request) -> httpx.Response:
self.request_body = json.loads(request.content)
return httpx.Response(200, json=dict(self.reply), request=request)
_ANTHROPIC_MESSAGE_PAYLOAD: Final = {
"id": "msg_turn_two",
"type": "message",
"role": "assistant",
"model": "claude-sonnet-4-6",
"content": [{"type": "text", "text": "14"}],
"stop_reason": "end_turn",
"stop_sequence": None,
"usage": {"input_tokens": 12, "output_tokens": 1},
}
@pytest.mark.asyncio
async def test_bridged_follow_up_turn_keeps_the_addressed_response_id_off_the_provider_body():
"""The proxy's ResponsesIDSecurity hook rewrites `previous_response_id` and keeps the
id the client addressed under `_litellm_addressed_response_id` in the same request
body, so internal retries re-authorize it. On a model without a native Responses
config that body is bridged into `litellm.acompletion` kwargs, and Azure AI Claude
answered `_litellm_addressed_response_id: Extra inputs are not permitted` (400) on
every follow-up turn. The key is LiteLLM-internal and must never reach the provider.
"""
provider: Final = _RecordingAnthropicHandler(_ANTHROPIC_MESSAGE_PAYLOAD)
client: Final = AsyncHTTPHandler()
client.client = httpx.AsyncClient(transport=httpx.MockTransport(provider))
response = await litellm.aresponses(
model="azure_ai/claude-sonnet-4-6",
api_base="https://fake-foundry-resource.services.ai.azure.com",
api_key="fake-api-key",
input="Double it",
previous_response_id="resp_turn_one",
client=client,
**{ADDRESSED_RESPONSE_ID_FIELD: "resp_turn_one"},
)
assert provider.request_body is not None, "the bridged turn never reached the provider"
assert ADDRESSED_RESPONSE_ID_FIELD not in provider.request_body, (
f"the addressed response id reached the provider body: {sorted(provider.request_body)}"
)
assert isinstance(response, ResponsesAPIResponse)
assert [item.type for item in response.output] == ["message"]

View file

@ -46,6 +46,7 @@ from litellm.types.utils import (
PromptTokensDetailsWrapper,
StreamingChoices,
Usage,
ADDRESSED_RESPONSE_ID_FIELD,
all_litellm_params,
bedrock_batch_litellm_params,
)
@ -4787,6 +4788,27 @@ def test_get_litellm_params_keys_never_reach_the_provider():
)
def test_addressed_response_id_never_reaches_the_provider():
"""The ResponsesIDSecurity hook keeps the id a client addressed under
`_litellm_addressed_response_id` in the request body so internal retries re-authorize
it. A bridged Responses call (no native Responses config, e.g. azure_ai Claude)
forwards that body as `completion()` kwargs, and the provider rejects the unknown
key: `_litellm_addressed_response_id: Extra inputs are not permitted`, a 400 on
every follow-up turn that carries `previous_response_id`.
"""
kwargs = {
"a_real_provider_specific_param": 1,
ADDRESSED_RESPONSE_ID_FIELD: "resp_addressed-by-the-client",
}
non_default = get_non_default_completion_params(kwargs)
assert non_default == {"a_real_provider_specific_param": 1}, (
"the addressed response id leaked into the provider params: "
f"{sorted(set(non_default) - {'a_real_provider_specific_param'})}"
)
def test_bedrock_batch_params_never_reach_the_provider():
"""A Bedrock managed-batch deployment carries aws_batch_role_arn / s3_* /
bedrock_tags in its litellm_params, and the same deployment also serves chat.