From 4ace5c8db3cbaf6ba498d35326fec73cba3db4c1 Mon Sep 17 00:00:00 2001 From: zhanghanduo Date: Sun, 16 Aug 2026 13:37:54 +0800 Subject: [PATCH] fix(apodex): route deep research through responses --- .../messages/handler.py | 4 +- litellm/llms/apodex/chat/transformation.py | 4 +- .../llms/apodex/responses/transformation.py | 82 ++++++++++++++++++- .../test_apodex_messages_transformation.py | 11 ++- .../test_apodex_responses_transformation.py | 46 +++++++++++ 5 files changed, 137 insertions(+), 10 deletions(-) diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py index c4b5cc628e2..128f891386c 100644 --- a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py +++ b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py @@ -37,9 +37,9 @@ from ..utils import is_reasoning_auto_summary_enabled from .interceptors import get_messages_interceptors from .utils import AnthropicMessagesRequestUtils, mock_response -# Providers that are routed directly to the OpenAI Responses API instead of +# Providers that are routed directly to a Responses API instead of # going through chat/completions. -_RESPONSES_API_PROVIDERS: Final = frozenset({"openai"}) +_RESPONSES_API_PROVIDERS: Final = frozenset({"apodex", "openai"}) def _should_route_to_responses_api(custom_llm_provider: str | None) -> bool: diff --git a/litellm/llms/apodex/chat/transformation.py b/litellm/llms/apodex/chat/transformation.py index 300e3e57991..38f3757658b 100644 --- a/litellm/llms/apodex/chat/transformation.py +++ b/litellm/llms/apodex/chat/transformation.py @@ -109,9 +109,7 @@ class ApodexChatConfig(OpenAIGPTConfig): # request body by the SDK, so it survives that drop. requested_extra_body: Final = renamed.get("extra_body") extra_body: Final = ( - requested_extra_body - if isinstance(requested_extra_body, Mapping) - else {} # mutable-ok: JSON request body + requested_extra_body if isinstance(requested_extra_body, Mapping) else {} # mutable-ok: JSON request body ) return { # mutable-ok: JSON request body **renamed, diff --git a/litellm/llms/apodex/responses/transformation.py b/litellm/llms/apodex/responses/transformation.py index 04ca2e55c74..a7820757bc9 100644 --- a/litellm/llms/apodex/responses/transformation.py +++ b/litellm/llms/apodex/responses/transformation.py @@ -14,20 +14,34 @@ Ref: https://platform.apodex.ai/docs/responses-api https://platform.apodex.ai/docs/models """ +from __future__ import annotations + from collections.abc import Mapping -from typing import Final +from time import time +from typing import TYPE_CHECKING, Final + +import httpx +from pydantic import TypeAdapter import litellm from litellm.llms.openai.responses.transformation import OpenAIResponsesAPIConfig -from litellm.types.llms.openai import ResponsesAPIOptionalRequestParams +from litellm.types.llms.openai import ( + ResponsesAPIOptionalRequestParams, + ResponsesAPIResponse, + ResponsesAPIStreamingResponse, +) from litellm.types.router import GenericLiteLLMParams from litellm.types.utils import LlmProviders -from ..common_utils import get_apodex_api_key, is_deep_research_model +from ..common_utils import get_apodex_api_base, get_apodex_api_key, is_deep_research_model + +if TYPE_CHECKING: + from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj # Rejected by the core models with HTTP 400: there is no server-side conversation # to resume and requests are always executed inline. _STATEFUL_PARAMS: Final = ("previous_response_id", "background") +_CANCEL_RESPONSE_ADAPTER: Final = TypeAdapter(dict[str, object]) class ApodexResponsesConfig(OpenAIResponsesAPIConfig): @@ -53,6 +67,68 @@ class ApodexResponsesConfig(OpenAIResponsesAPIConfig): "Authorization": f"Bearer {api_key}", } + def get_complete_url( + self, + api_base: str | None, + litellm_params: dict, # mutable-ok: matches the base-class signature + ) -> str: + resolved_base: Final = get_apodex_api_base(api_base).rstrip("/") + return f"{resolved_base}/responses" + + def transform_cancel_response_api_response( + self, + raw_response: httpx.Response, + logging_obj: LiteLLMLoggingObj, + ) -> ResponsesAPIResponse: + payload: Final = _CANCEL_RESPONSE_ADAPTER.validate_json(raw_response.content) + normalized_response: Final = httpx.Response( + status_code=raw_response.status_code, + headers=raw_response.headers, + json={ + **payload, + "created_at": payload.get("created_at", int(time())), + "output": payload.get("output", []), + }, + ) + return super().transform_cancel_response_api_response( + raw_response=normalized_response, + logging_obj=logging_obj, + ) + + def transform_streaming_response( + self, + model: str, + parsed_chunk: dict, # mutable-ok: matches the base-class signature + logging_obj: LiteLLMLoggingObj, + ) -> ResponsesAPIStreamingResponse: + swarm: Final = parsed_chunk.get("swarm") + swarm_data: Final = swarm.get("data") if isinstance(swarm, dict) else None + if ( + parsed_chunk.get("type") != "response.swarm.llm_delta" + or not isinstance(swarm_data, dict) + or swarm_data.get("channel") != "output_text" + or not isinstance(swarm_data.get("delta"), str) + ): + return super().transform_streaming_response( + model=model, + parsed_chunk=parsed_chunk, + logging_obj=logging_obj, + ) + + response_id: Final = str(parsed_chunk.get("response_id", "")) + return super().transform_streaming_response( + model=model, + parsed_chunk={ + "type": "response.output_text.delta", + "item_id": f"msg_{response_id}", + "output_index": 0, + "content_index": 0, + "delta": swarm_data["delta"], + "sequence_number": parsed_chunk.get("sequence_number", 0), + }, + logging_obj=logging_obj, + ) + def get_supported_openai_params(self, model: str) -> list: # mutable-ok: matches the base-class signature inherited: Final = super().get_supported_openai_params(model) if is_deep_research_model(model): diff --git a/tests/test_litellm/llms/apodex/test_apodex_messages_transformation.py b/tests/test_litellm/llms/apodex/test_apodex_messages_transformation.py index 9d8c787c75b..09427b31a2f 100644 --- a/tests/test_litellm/llms/apodex/test_apodex_messages_transformation.py +++ b/tests/test_litellm/llms/apodex/test_apodex_messages_transformation.py @@ -58,10 +58,17 @@ class TestNativePassthroughRouting: @pytest.mark.parametrize("model", DEEP_RESEARCH_MODELS) def test_deep_research_models_fall_back_to_translation(self, model: str): - """No native config means LiteLLM translates to chat completions, which works, - instead of forwarding to a path Apodex does not serve for these tiers.""" + """No native config means LiteLLM uses a protocol translation instead of + forwarding to a path Apodex does not serve for these tiers.""" assert _messages_config(model) is None + def test_deep_research_translation_uses_responses_api(self): + from litellm.llms.anthropic.experimental_pass_through.messages.handler import ( + _should_route_to_responses_api, + ) + + assert _should_route_to_responses_api("apodex") is True + class TestNativePassthroughRequest: def test_url_targets_the_native_messages_path(self): diff --git a/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py b/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py index d343f4553c8..44eb0cb4b1e 100644 --- a/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py +++ b/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py @@ -6,6 +6,7 @@ Research tiers keep server-side state, so the parameter contract is keyed off the model rather than applied provider-wide. """ +import httpx import pytest import litellm @@ -80,6 +81,17 @@ class TestConfigSelection: def test_request_targets_the_apodex_responses_url(self): assert _capture(model=CORE_MODEL, input="hi")["url"] == "https://api.apodex.ai/v1/responses" + def test_polling_without_model_resolution_targets_apodex(self): + config = _responses_config("apodex-1-1-deep-research") + assert config.get_complete_url(api_base=None, litellm_params={}) == "https://api.apodex.ai/v1/responses" + + def test_polling_honours_an_explicit_api_base(self): + config = _responses_config("apodex-1-1-deep-research") + assert ( + config.get_complete_url(api_base="https://gateway.apodex.test/v1/", litellm_params={}) + == "https://gateway.apodex.test/v1/responses" + ) + def test_request_honours_an_api_base_override(self, monkeypatch: pytest.MonkeyPatch): monkeypatch.setenv("APODEX_API_BASE", "https://env.apodex.test/v1") assert _capture(model=CORE_MODEL, input="hi")["url"] == "https://env.apodex.test/v1/responses" @@ -175,3 +187,37 @@ class TestDeepResearchKeepsState: ) assert "background" in supported assert "previous_response_id" in supported + + def test_minimal_cancel_response_is_normalized(self): + config = _responses_config("apodex-1-1-deep-research") + response = config.transform_cancel_response_api_response( + raw_response=httpx.Response( + 200, + json={"id": "resp_1", "object": "response", "status": "cancelled"}, + ), + logging_obj=None, + ) + + assert response.id == "resp_1" + assert response.status == "cancelled" + assert response.output == [] + assert response.created_at > 0 + + def test_deep_research_output_delta_is_normalized(self): + config = _responses_config("apodex-1-1-deep-research") + event = config.transform_streaming_response( + model="apodex-1-1-deep-research", + parsed_chunk={ + "type": "response.swarm.llm_delta", + "response_id": "w_123", + "sequence_number": 12, + "swarm": { + "agent_id": "reporter", + "data": {"channel": "output_text", "delta": "final answer"}, + }, + }, + logging_obj=None, + ) + assert event.type == "response.output_text.delta" + assert event.item_id == "msg_w_123" + assert event.delta == "final answer"