Forward custom_llm_provider through the Responses API bridge (Fixes #28505) (#28575)

* Forward custom_llm_provider through the Responses API bridge (Fixes #28505)

When a Chat Completions request to a GPT-5.4+ model contains both
`tools` and `reasoning_effort`, `completion()` auto-routes through
`responses_api_bridge`. The bridge handler called
`litellm.responses()` / `litellm.aresponses()` without forwarding the
already-resolved `custom_llm_provider`, so the downstream call
re-invoked `get_llm_provider()` with `custom_llm_provider=None` and
stripped a second provider prefix from a `provider/provider/model`
deployment string.

For a deployment configured as `openai/openai/openai/gpt-5.5`,
the bridge flow sent `openai/gpt-5.5` to the upstream API instead of
the correct `openai/openai/gpt-5.5`. Upstream APIs that enforce
model-name allow-lists rejected this as `key_model_access_denied`.

Fix: pass the locally-resolved `custom_llm_provider` into both the
sync `responses()` and async `aresponses()` calls so the downstream
`_resolve_model_provider_for_responses` sees an explicit provider
and skips the second prefix-strip.

New regression test
`tests/test_litellm/completion_extras/test_responses_bridge_provider_propagation.py`
pins both call sites: each must forward `custom_llm_provider`.

* fix(28505): set custom_llm_provider on request_data instead of as duplicate kwarg

Greptile flagged that the previous patch passed custom_llm_provider as an
explicit kwarg to responses()/aresponses() while request_data already
carried it via the spread of sanitized_litellm_params, which would raise
TypeError: got multiple values for keyword argument on every real bridge
call.

Switches to assigning request_data['custom_llm_provider'] before the call
so the resolved provider wins over whatever sanitized_litellm_params spread
in, without duplicating the kwarg.

Updates the regression test to seed request_data with a sentinel
custom_llm_provider so it actually exercises the overwrite path (the
previous test mocked transform_request with a minimal dict and never hit
the conflict).

* chore: trigger shin-agent re-eval on retargeted staging base

* chore: trigger shin-agent re-eval against updated Greptile state
This commit is contained in:
Filippo Menghi 2026-06-01 13:02:17 +02:00 • committed by GitHub
parent 2c144fcfbc
commit a4bb402cf3
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
3 changed files with 131 additions and 0 deletions

View file

@ -182,6 +182,14 @@ class ResponsesToCompletionBridgeHandler:
client=kwargs.get("client"),
)
# Pin the resolved provider so `responses()` doesn't re-run
# `get_llm_provider()` on the model string and strip a second
# provider prefix (see GitHub issue #28505). request_data already
# carries `custom_llm_provider` via the spread of
# `sanitized_litellm_params`; overwriting it on the dict (rather
# than adding an explicit kwarg) avoids the duplicate-keyword
# TypeError that would otherwise fire on the real bridge path.
request_data["custom_llm_provider"] = custom_llm_provider
result = responses(
**request_data,
)
@ -268,6 +276,13 @@ class ResponsesToCompletionBridgeHandler:
except Exception as e:
raise e
# Pin the resolved provider so `aresponses()` doesn't re-run
# `get_llm_provider()` on the model string and strip a second
# provider prefix (see GitHub issue #28505). Set on request_data
# rather than passed as a separate kwarg to avoid the duplicate-
# keyword TypeError when `sanitized_litellm_params` already
# carries `custom_llm_provider`.
request_data["custom_llm_provider"] = custom_llm_provider
result = await aresponses(
**request_data,
aresponses=True,

View file

@ -0,0 +1,116 @@
"""
Regression test for https://github.com/BerriAI/litellm/issues/28505 -
the Responses API bridge double-strips the provider prefix from the
model name when a Chat Completions request has both `tools` and
`reasoning_effort`.
Root cause: the bridge handler called `litellm.responses()` /
`litellm.aresponses()` without passing the already-resolved
`custom_llm_provider`. The downstream call then re-invoked
`get_llm_provider()` with `custom_llm_provider=None`, which stripped
a second provider prefix from a `provider/provider/model` deployment
string.
This test pins both the sync and async bridge handler call sites:
the resolved `custom_llm_provider` must be forwarded to the underlying
`responses` / `aresponses` call so the provider isn't re-detected.
"""
from unittest.mock import MagicMock, patch
import pytest
from litellm.completion_extras.litellm_responses_transformation.handler import (
ResponsesToCompletionBridgeHandler,
)
def _validated_kwargs():
return {
"model": "openai/openai/openai/gpt-5.5",
"messages": [{"role": "user", "content": "hi"}],
"optional_params": {},
"litellm_params": {},
"headers": {},
"model_response": MagicMock(),
"logging_obj": MagicMock(),
"custom_llm_provider": "openai",
}
def test_sync_completion_forwards_custom_llm_provider():
handler = ResponsesToCompletionBridgeHandler()
handler.transformation_handler = MagicMock()
handler.transformation_handler.transform_request.return_value = {
"model": "openai/openai/openai/gpt-5.5",
"input": [],
# `_build_sanitized_litellm_params` spreads `custom_llm_provider` from
# `litellm_params` into request_data on the real bridge path. Seed
# it here so the test exercises the overwrite (not an explicit kwarg
# that would TypeError against an already-present key).
"custom_llm_provider": "should-be-overwritten",
}
handler.transformation_handler.transform_response.return_value = (
_validated_kwargs()["model_response"]
)
with (
patch.object(
handler, "validate_input_kwargs", return_value=_validated_kwargs()
),
patch(
"litellm.responses",
return_value=MagicMock(spec=[]),
) as mock_responses,
):
# The handler routes ResponsesAPIResponse through transform_response.
# We just want to verify the kwargs going INTO responses().
try:
handler.completion(acompletion=False)
except Exception:
# Downstream handling (transform_response, type checks) is not
# the subject of this test.
pass
assert mock_responses.called
kwargs = mock_responses.call_args.kwargs
assert kwargs.get("custom_llm_provider") == "openai", (
"sync bridge must forward custom_llm_provider to litellm.responses() "
"so the downstream get_llm_provider() call does not re-strip the "
"provider prefix on a provider/provider/model deployment string"
)
@pytest.mark.asyncio
async def test_async_completion_forwards_custom_llm_provider():
handler = ResponsesToCompletionBridgeHandler()
handler.transformation_handler = MagicMock()
handler.transformation_handler.transform_request.return_value = {
"model": "openai/openai/openai/gpt-5.5",
"input": [],
# `_build_sanitized_litellm_params` spreads `custom_llm_provider` from
# `litellm_params` into request_data on the real bridge path. Seed
# it here so the test exercises the overwrite (not an explicit kwarg
# that would TypeError against an already-present key).
"custom_llm_provider": "should-be-overwritten",
}
async def _fake_aresponses(**kwargs):
_fake_aresponses.kwargs = kwargs
return MagicMock(spec=[])
_fake_aresponses.kwargs = {}
with (
patch.object(
handler, "validate_input_kwargs", return_value=_validated_kwargs()
),
patch("litellm.aresponses", _fake_aresponses),
):
try:
await handler.acompletion()
except Exception:
pass
assert _fake_aresponses.kwargs.get("custom_llm_provider") == "openai", (
"async bridge must forward custom_llm_provider to litellm.aresponses() "
"so the downstream get_llm_provider() call does not re-strip the "
"provider prefix on a provider/provider/model deployment string"
)