From 67662565e8bb95a1ea054743d1bea37e43d35e7c Mon Sep 17 00:00:00 2001 From: Prathamesh Jadhav <55660103+lollinng@users.noreply.github.com> Date: Wed, 17 Jun 2026 16:56:07 +0530 Subject: [PATCH] feat(dashscope): add Responses API support (#30286) * feat(dashscope): add Responses API support DashScope's OpenAI-compatible endpoint serves /responses, so register a DashScopeResponsesAPIConfig that routes dashscope/* responses calls to {api_base}/responses without rewriting the upstream model id, instead of falling back to the chat-completions -> responses emulation pipeline. Closes #29780 * feat(dashscope): mark responses API as not supporting native websocket Matches the hosted_vllm/perplexity/openrouter responses configs, which all override supports_native_websocket() to False since the OpenAI-compatible endpoint has no native wss:// responses transport. --------- Co-authored-by: Sameer Kankute --- litellm/__init__.py | 2 + litellm/_lazy_imports_registry.py | 5 + litellm/llms/dashscope/responses/__init__.py | 0 .../dashscope/responses/transformation.py | 70 +++++++ litellm/utils.py | 2 + .../llms/dashscope/responses/__init__.py | 0 ...test_dashscope_responses_transformation.py | 178 ++++++++++++++++++ 7 files changed, 257 insertions(+) create mode 100644 litellm/llms/dashscope/responses/__init__.py create mode 100644 litellm/llms/dashscope/responses/transformation.py create mode 100644 tests/test_litellm/llms/dashscope/responses/__init__.py create mode 100644 tests/test_litellm/llms/dashscope/responses/test_dashscope_responses_transformation.py diff --git a/litellm/__init__.py b/litellm/__init__.py index 9b48e7382f0..8daf54d1c4b 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -1980,6 +1980,8 @@ if TYPE_CHECKING: from .llms.dashscope.rerank.transformation import ( DashScopeRerankConfig as DashScopeRerankConfig, ) + from .llms.dashscope.responses.transformation import ( + DashScopeResponsesAPIConfig as DashScopeResponsesAPIConfig, from .llms.modelscope.chat.transformation import ( ModelScopeChatConfig as ModelScopeChatConfig, ) diff --git a/litellm/_lazy_imports_registry.py b/litellm/_lazy_imports_registry.py index e653b40fd04..d8d72dfab57 100644 --- a/litellm/_lazy_imports_registry.py +++ b/litellm/_lazy_imports_registry.py @@ -306,6 +306,7 @@ LLM_CONFIG_NAMES = ( "GigaChatConfig", "GigaChatEmbeddingConfig", "DashScopeChatConfig", + "DashScopeResponsesAPIConfig", "ModelScopeChatConfig", "MoonshotChatConfig", "DockerModelRunnerChatConfig", @@ -1162,6 +1163,10 @@ _LLM_CONFIGS_IMPORT_MAP = { ".llms.dashscope.chat.transformation", "DashScopeChatConfig", ), + "DashScopeResponsesAPIConfig": ( + ".llms.dashscope.responses.transformation", + "DashScopeResponsesAPIConfig", + ), "ModelScopeChatConfig": ( ".llms.modelscope.chat.transformation", "ModelScopeChatConfig", diff --git a/litellm/llms/dashscope/responses/__init__.py b/litellm/llms/dashscope/responses/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/litellm/llms/dashscope/responses/transformation.py b/litellm/llms/dashscope/responses/transformation.py new file mode 100644 index 00000000000..3fd56db9444 --- /dev/null +++ b/litellm/llms/dashscope/responses/transformation.py @@ -0,0 +1,70 @@ +""" +Responses API transformation for DashScope (Alibaba Qwen) provider. + +DashScope exposes an OpenAI-compatible endpoint that serves the `/responses` +route, so this config enables direct routing to `{api_base}/responses` without +rewriting the upstream model id, instead of falling back to the chat +completions -> responses conversion pipeline. +""" + +from typing import Optional + +from litellm.llms.openai.responses.transformation import OpenAIResponsesAPIConfig +from litellm.secret_managers.main import get_secret_str +from litellm.types.router import GenericLiteLLMParams +from litellm.types.utils import LlmProviders + +DEFAULT_DASHSCOPE_API_BASE = "https://dashscope.aliyuncs.com/compatible-mode/v1" + + +class DashScopeResponsesAPIConfig(OpenAIResponsesAPIConfig): + """ + Configuration for DashScope Responses API support. + + Extends OpenAI's config since DashScope follows the OpenAI API spec, but + resolves the base URL/key from DASHSCOPE_API_BASE / DASHSCOPE_API_KEY and + routes to the DashScope compatible-mode endpoint. + """ + + @property + def custom_llm_provider(self) -> LlmProviders: + return LlmProviders.DASHSCOPE + + def validate_environment( + self, + headers: dict, + model: str, + litellm_params: Optional[GenericLiteLLMParams], + ) -> dict: + litellm_params = litellm_params or GenericLiteLLMParams() + api_key = litellm_params.api_key or get_secret_str("DASHSCOPE_API_KEY") + if api_key is None: + raise ValueError( + "DashScope API key not set for responses API. " + "Set via api_key parameter or DASHSCOPE_API_KEY environment variable" + ) + headers.update( + { + "Authorization": f"Bearer {api_key}", + } + ) + return headers + + def get_complete_url( + self, + api_base: Optional[str], + litellm_params: dict, + ) -> str: + api_base = ( + api_base + or get_secret_str("DASHSCOPE_API_BASE") + or DEFAULT_DASHSCOPE_API_BASE + ).rstrip("/") + + if api_base.endswith("/v1"): + return f"{api_base}/responses" + + return f"{api_base}/v1/responses" + + def supports_native_websocket(self) -> bool: + return False diff --git a/litellm/utils.py b/litellm/utils.py index 30b5691a140..58a08c06c88 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -9069,6 +9069,8 @@ class ProviderConfigManager: return litellm.OpenRouterResponsesAPIConfig() elif litellm.LlmProviders.HOSTED_VLLM == provider: return litellm.HostedVLLMResponsesAPIConfig() + elif litellm.LlmProviders.DASHSCOPE == provider: + return litellm.DashScopeResponsesAPIConfig() elif litellm.LlmProviders.BEDROCK_MANTLE == provider: # Mantle serves Responses on two upstream paths. A model takes the # /openai/v1/responses path when its price-map entry declares diff --git a/tests/test_litellm/llms/dashscope/responses/__init__.py b/tests/test_litellm/llms/dashscope/responses/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/tests/test_litellm/llms/dashscope/responses/test_dashscope_responses_transformation.py b/tests/test_litellm/llms/dashscope/responses/test_dashscope_responses_transformation.py new file mode 100644 index 00000000000..5eea5810428 --- /dev/null +++ b/tests/test_litellm/llms/dashscope/responses/test_dashscope_responses_transformation.py @@ -0,0 +1,178 @@ +""" +Tests for DashScope (Alibaba Qwen) Responses API support. + +Feature: https://github.com/BerriAI/litellm/issues/29780 +DashScope exposes an OpenAI-compatible /responses endpoint. These tests pin the +two guarantees the issue asks for: dashscope/* responses calls route to +{api_base}/responses, and the upstream model id is passed through unchanged. +""" + +import json +import os +import sys +from unittest.mock import MagicMock, patch + +import pytest + +sys.path.insert(0, os.path.abspath("../../../../..")) + +import litellm +from litellm.llms.dashscope.responses.transformation import ( + DashScopeResponsesAPIConfig, +) +from litellm.types.router import GenericLiteLLMParams +from litellm.types.utils import LlmProviders +from litellm.utils import ProviderConfigManager + +DASHSCOPE_RESPONSES_URL = "https://dashscope.aliyuncs.com/compatible-mode/v1/responses" + + +def _make_mock_responses_api_response() -> dict: + return { + "id": "resp-test123", + "object": "response", + "created_at": 1234567890, + "model": "qwen-max", + "output": [ + { + "type": "message", + "id": "msg-test123", + "status": "completed", + "role": "assistant", + "content": [ + { + "type": "output_text", + "text": "Hello from Qwen", + "annotations": [], + } + ], + } + ], + "status": "completed", + "usage": {"input_tokens": 10, "output_tokens": 20, "total_tokens": 30}, + } + + +def _make_mock_http_client(response_body: dict) -> MagicMock: + mock_client = MagicMock() + mock_response = MagicMock() + mock_response.status_code = 200 + mock_response.headers = {"content-type": "application/json"} + mock_response.json.return_value = response_body + mock_response.text = json.dumps(response_body) + mock_client.post.return_value = mock_response + return mock_client + + +def _extract_posted_url_and_body(mock_client: MagicMock): + call = mock_client.post.call_args + url = call.kwargs.get("url") or (call.args[0] if call.args else None) + raw_body = call.kwargs.get("data") + if raw_body is None: + raw_body = call.kwargs.get("json") + body = json.loads(raw_body) if isinstance(raw_body, str) else raw_body + return url, body + + +def test_dashscope_responses_routes_to_responses_endpoint_without_model_rewrite(): + """End-to-end: dashscope/* responses call hits {base}/responses and sends the + upstream model id ('qwen-max') unchanged. This is the core of issue #29780.""" + mock_client = _make_mock_http_client(_make_mock_responses_api_response()) + + with patch( + "litellm.llms.custom_httpx.llm_http_handler._get_httpx_client", + return_value=mock_client, + ): + response = litellm.responses( + model="dashscope/qwen-max", + input="Hello, how are you?", + api_key="test-key", + ) + + url, body = _extract_posted_url_and_body(mock_client) + assert url == DASHSCOPE_RESPONSES_URL + assert body["model"] == "qwen-max" + + from litellm.types.llms.openai import ResponsesAPIResponse + + assert isinstance(response, ResponsesAPIResponse) + assert response.output[0].content[0].text == "Hello from Qwen" # type: ignore[union-attr] + + +def test_dashscope_provider_config_registration(): + """ProviderConfigManager must return the DashScope responses config so the call + routes natively instead of falling back to chat-completions emulation.""" + config = ProviderConfigManager.get_provider_responses_api_config( + model="dashscope/qwen-max", + provider=LlmProviders.DASHSCOPE, + ) + + assert isinstance(config, DashScopeResponsesAPIConfig) + assert config.custom_llm_provider == LlmProviders.DASHSCOPE + + +def test_dashscope_responses_api_url(): + config = DashScopeResponsesAPIConfig() + + assert config.get_complete_url(api_base=None, litellm_params={}) == ( + DASHSCOPE_RESPONSES_URL + ) + assert ( + config.get_complete_url( + api_base="https://dashscope.aliyuncs.com/compatible-mode/v1/", + litellm_params={}, + ) + == DASHSCOPE_RESPONSES_URL + ) + assert ( + config.get_complete_url( + api_base="https://proxy.internal/api", litellm_params={} + ) + == "https://proxy.internal/api/v1/responses" + ) + + +def test_dashscope_responses_api_url_uses_env_base(monkeypatch): + monkeypatch.setenv("DASHSCOPE_API_BASE", "https://intl.example/compatible-mode/v1") + config = DashScopeResponsesAPIConfig() + + assert config.get_complete_url(api_base=None, litellm_params={}) == ( + "https://intl.example/compatible-mode/v1/responses" + ) + + +def test_dashscope_validate_environment_sets_bearer_from_param(): + config = DashScopeResponsesAPIConfig() + + headers = config.validate_environment( + headers={}, + model="qwen-max", + litellm_params=GenericLiteLLMParams(api_key="sk-dashscope"), + ) + + assert headers["Authorization"] == "Bearer sk-dashscope" + + +def test_dashscope_validate_environment_reads_env_key(monkeypatch): + monkeypatch.setenv("DASHSCOPE_API_KEY", "sk-from-env") + config = DashScopeResponsesAPIConfig() + + headers = config.validate_environment( + headers={}, model="qwen-max", litellm_params=GenericLiteLLMParams() + ) + + assert headers["Authorization"] == "Bearer sk-from-env" + + +def test_dashscope_validate_environment_requires_api_key(monkeypatch): + monkeypatch.delenv("DASHSCOPE_API_KEY", raising=False) + config = DashScopeResponsesAPIConfig() + + with pytest.raises(ValueError, match="DashScope API key not set"): + config.validate_environment( + headers={}, model="qwen-max", litellm_params=GenericLiteLLMParams() + ) + + +def test_dashscope_does_not_support_native_websocket(): + assert DashScopeResponsesAPIConfig().supports_native_websocket() is False