mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-29 01:42:19 +00:00
* Forward custom_llm_provider through the Responses API bridge (Fixes #28505) When a Chat Completions request to a GPT-5.4+ model contains both `tools` and `reasoning_effort`, `completion()` auto-routes through `responses_api_bridge`. The bridge handler called `litellm.responses()` / `litellm.aresponses()` without forwarding the already-resolved `custom_llm_provider`, so the downstream call re-invoked `get_llm_provider()` with `custom_llm_provider=None` and stripped a second provider prefix from a `provider/provider/model` deployment string. For a deployment configured as `openai/openai/openai/gpt-5.5`, the bridge flow sent `openai/gpt-5.5` to the upstream API instead of the correct `openai/openai/gpt-5.5`. Upstream APIs that enforce model-name allow-lists rejected this as `key_model_access_denied`. Fix: pass the locally-resolved `custom_llm_provider` into both the sync `responses()` and async `aresponses()` calls so the downstream `_resolve_model_provider_for_responses` sees an explicit provider and skips the second prefix-strip. New regression test `tests/test_litellm/completion_extras/test_responses_bridge_provider_propagation.py` pins both call sites: each must forward `custom_llm_provider`. * fix(28505): set custom_llm_provider on request_data instead of as duplicate kwarg Greptile flagged that the previous patch passed custom_llm_provider as an explicit kwarg to responses()/aresponses() while request_data already carried it via the spread of sanitized_litellm_params, which would raise TypeError: got multiple values for keyword argument on every real bridge call. Switches to assigning request_data['custom_llm_provider'] before the call so the resolved provider wins over whatever sanitized_litellm_params spread in, without duplicating the kwarg. Updates the regression test to seed request_data with a sentinel custom_llm_provider so it actually exercises the overwrite path (the previous test mocked transform_request with a minimal dict and never hit the conflict). * chore: trigger shin-agent re-eval on retargeted staging base * chore: trigger shin-agent re-eval against updated Greptile state
This commit is contained in:
parent
2c144fcfbc
commit
a4bb402cf3
3 changed files with 131 additions and 0 deletions
|
|
@ -182,6 +182,14 @@ class ResponsesToCompletionBridgeHandler:
|
|||
client=kwargs.get("client"),
|
||||
)
|
||||
|
||||
# Pin the resolved provider so `responses()` doesn't re-run
|
||||
# `get_llm_provider()` on the model string and strip a second
|
||||
# provider prefix (see GitHub issue #28505). request_data already
|
||||
# carries `custom_llm_provider` via the spread of
|
||||
# `sanitized_litellm_params`; overwriting it on the dict (rather
|
||||
# than adding an explicit kwarg) avoids the duplicate-keyword
|
||||
# TypeError that would otherwise fire on the real bridge path.
|
||||
request_data["custom_llm_provider"] = custom_llm_provider
|
||||
result = responses(
|
||||
**request_data,
|
||||
)
|
||||
|
|
@ -268,6 +276,13 @@ class ResponsesToCompletionBridgeHandler:
|
|||
except Exception as e:
|
||||
raise e
|
||||
|
||||
# Pin the resolved provider so `aresponses()` doesn't re-run
|
||||
# `get_llm_provider()` on the model string and strip a second
|
||||
# provider prefix (see GitHub issue #28505). Set on request_data
|
||||
# rather than passed as a separate kwarg to avoid the duplicate-
|
||||
# keyword TypeError when `sanitized_litellm_params` already
|
||||
# carries `custom_llm_provider`.
|
||||
request_data["custom_llm_provider"] = custom_llm_provider
|
||||
result = await aresponses(
|
||||
**request_data,
|
||||
aresponses=True,
|
||||
|
|
|
|||
0
tests/test_litellm/completion_extras/__init__.py
Normal file
0
tests/test_litellm/completion_extras/__init__.py
Normal file
|
|
@ -0,0 +1,116 @@
|
|||
"""
|
||||
Regression test for https://github.com/BerriAI/litellm/issues/28505 -
|
||||
the Responses API bridge double-strips the provider prefix from the
|
||||
model name when a Chat Completions request has both `tools` and
|
||||
`reasoning_effort`.
|
||||
|
||||
Root cause: the bridge handler called `litellm.responses()` /
|
||||
`litellm.aresponses()` without passing the already-resolved
|
||||
`custom_llm_provider`. The downstream call then re-invoked
|
||||
`get_llm_provider()` with `custom_llm_provider=None`, which stripped
|
||||
a second provider prefix from a `provider/provider/model` deployment
|
||||
string.
|
||||
|
||||
This test pins both the sync and async bridge handler call sites:
|
||||
the resolved `custom_llm_provider` must be forwarded to the underlying
|
||||
`responses` / `aresponses` call so the provider isn't re-detected.
|
||||
"""
|
||||
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from litellm.completion_extras.litellm_responses_transformation.handler import (
|
||||
ResponsesToCompletionBridgeHandler,
|
||||
)
|
||||
|
||||
|
||||
def _validated_kwargs():
|
||||
return {
|
||||
"model": "openai/openai/openai/gpt-5.5",
|
||||
"messages": [{"role": "user", "content": "hi"}],
|
||||
"optional_params": {},
|
||||
"litellm_params": {},
|
||||
"headers": {},
|
||||
"model_response": MagicMock(),
|
||||
"logging_obj": MagicMock(),
|
||||
"custom_llm_provider": "openai",
|
||||
}
|
||||
|
||||
|
||||
def test_sync_completion_forwards_custom_llm_provider():
|
||||
handler = ResponsesToCompletionBridgeHandler()
|
||||
handler.transformation_handler = MagicMock()
|
||||
handler.transformation_handler.transform_request.return_value = {
|
||||
"model": "openai/openai/openai/gpt-5.5",
|
||||
"input": [],
|
||||
# `_build_sanitized_litellm_params` spreads `custom_llm_provider` from
|
||||
# `litellm_params` into request_data on the real bridge path. Seed
|
||||
# it here so the test exercises the overwrite (not an explicit kwarg
|
||||
# that would TypeError against an already-present key).
|
||||
"custom_llm_provider": "should-be-overwritten",
|
||||
}
|
||||
handler.transformation_handler.transform_response.return_value = (
|
||||
_validated_kwargs()["model_response"]
|
||||
)
|
||||
with (
|
||||
patch.object(
|
||||
handler, "validate_input_kwargs", return_value=_validated_kwargs()
|
||||
),
|
||||
patch(
|
||||
"litellm.responses",
|
||||
return_value=MagicMock(spec=[]),
|
||||
) as mock_responses,
|
||||
):
|
||||
# The handler routes ResponsesAPIResponse through transform_response.
|
||||
# We just want to verify the kwargs going INTO responses().
|
||||
try:
|
||||
handler.completion(acompletion=False)
|
||||
except Exception:
|
||||
# Downstream handling (transform_response, type checks) is not
|
||||
# the subject of this test.
|
||||
pass
|
||||
assert mock_responses.called
|
||||
kwargs = mock_responses.call_args.kwargs
|
||||
assert kwargs.get("custom_llm_provider") == "openai", (
|
||||
"sync bridge must forward custom_llm_provider to litellm.responses() "
|
||||
"so the downstream get_llm_provider() call does not re-strip the "
|
||||
"provider prefix on a provider/provider/model deployment string"
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_async_completion_forwards_custom_llm_provider():
|
||||
handler = ResponsesToCompletionBridgeHandler()
|
||||
handler.transformation_handler = MagicMock()
|
||||
handler.transformation_handler.transform_request.return_value = {
|
||||
"model": "openai/openai/openai/gpt-5.5",
|
||||
"input": [],
|
||||
# `_build_sanitized_litellm_params` spreads `custom_llm_provider` from
|
||||
# `litellm_params` into request_data on the real bridge path. Seed
|
||||
# it here so the test exercises the overwrite (not an explicit kwarg
|
||||
# that would TypeError against an already-present key).
|
||||
"custom_llm_provider": "should-be-overwritten",
|
||||
}
|
||||
|
||||
async def _fake_aresponses(**kwargs):
|
||||
_fake_aresponses.kwargs = kwargs
|
||||
return MagicMock(spec=[])
|
||||
|
||||
_fake_aresponses.kwargs = {}
|
||||
|
||||
with (
|
||||
patch.object(
|
||||
handler, "validate_input_kwargs", return_value=_validated_kwargs()
|
||||
),
|
||||
patch("litellm.aresponses", _fake_aresponses),
|
||||
):
|
||||
try:
|
||||
await handler.acompletion()
|
||||
except Exception:
|
||||
pass
|
||||
assert _fake_aresponses.kwargs.get("custom_llm_provider") == "openai", (
|
||||
"async bridge must forward custom_llm_provider to litellm.aresponses() "
|
||||
"so the downstream get_llm_provider() call does not re-strip the "
|
||||
"provider prefix on a provider/provider/model deployment string"
|
||||
)
|
||||
Loading…
Add table
Reference in a new issue