diff --git a/litellm/litellm_core_utils/get_llm_provider_logic.py b/litellm/litellm_core_utils/get_llm_provider_logic.py index ac1faeda645..4ea544b5baa 100644 --- a/litellm/litellm_core_utils/get_llm_provider_logic.py +++ b/litellm/litellm_core_utils/get_llm_provider_logic.py @@ -351,7 +351,6 @@ def get_llm_provider( dynamic_api_key = get_secret_str("META_API_KEY") elif endpoint == litellm.ApodexChatConfig.API_BASE_URL: custom_llm_provider = "apodex" # rebind-ok: dispatch chain resolves in place - # rebind-ok: dispatch chain resolves in place dynamic_api_key = litellm.ApodexChatConfig.get_api_key() if api_base is not None and not isinstance(api_base, str): diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py index 128f891386c..bafab2c9244 100644 --- a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py +++ b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py @@ -529,7 +529,8 @@ def anthropic_messages_handler( anthropic_messages_provider_config = OpenAILikeAnthropicMessagesConfig() if anthropic_messages_provider_config is None: - # Route to Responses API for OpenAI / Azure, chat/completions for everything else. + # Route to a Responses API for the providers that serve one, chat/completions + # for everything else. _shared_kwargs: Final = dict( max_tokens=max_tokens, messages=messages, diff --git a/litellm/llms/apodex/chat/transformation.py b/litellm/llms/apodex/chat/transformation.py index 38f3757658b..4db9bc9fac1 100644 --- a/litellm/llms/apodex/chat/transformation.py +++ b/litellm/llms/apodex/chat/transformation.py @@ -1,12 +1,14 @@ """ Apodex chat completions — OpenAI-compatible, with two provider quirks: -- `stream` defaults to true upstream, so a non-streaming call has to say so - explicitly or Apodex answers with SSE that a plain call cannot parse +- the Deep Research tiers default `stream` to true, so a non-streaming call has + to say so explicitly or Apodex answers with SSE that a plain call cannot + parse. The core models follow OpenAI and default it to false - the Deep Research tiers ignore sampling parameters and reject OpenAI-style tools; only the core models take them Ref: https://platform.apodex.ai/docs/chat-completions + https://platform.apodex.ai/docs/models """ from collections.abc import Mapping @@ -104,9 +106,10 @@ class ApodexChatConfig(OpenAIGPTConfig): if renamed.get("stream"): return renamed - # The OpenAI SDK drops `stream` from the body when it is false, which would - # leave Apodex on its streaming default. extra_body is merged into the - # request body by the SDK, so it survives that drop. + # The OpenAI chat handler pops `stream` out of the params it forwards + # (litellm/llms/openai/openai.py), which would leave the Deep Research tiers + # on their streaming default. extra_body is merged into the request body + # further down, so it survives that pop. requested_extra_body: Final = renamed.get("extra_body") extra_body: Final = ( requested_extra_body if isinstance(requested_extra_body, Mapping) else {} # mutable-ok: JSON request body diff --git a/litellm/llms/apodex/responses/transformation.py b/litellm/llms/apodex/responses/transformation.py index a7820757bc9..e8014905846 100644 --- a/litellm/llms/apodex/responses/transformation.py +++ b/litellm/llms/apodex/responses/transformation.py @@ -8,7 +8,8 @@ provider-wide: - core models are a stateless subset: `store` is forced to false, and `previous_response_id` or `background` come back as HTTP 400 - the Deep Research tiers keep server-side state, so they take all three -- both default `stream` to true, so a non-streaming call has to say so +- the Deep Research tiers default `stream` to true, so a non-streaming call has + to say so; pinning it for the core models too keeps one code path Ref: https://platform.apodex.ai/docs/responses-api https://platform.apodex.ai/docs/models @@ -21,14 +22,13 @@ from time import time from typing import TYPE_CHECKING, Final import httpx -from pydantic import TypeAdapter +from pydantic import TypeAdapter, ValidationError import litellm from litellm.llms.openai.responses.transformation import OpenAIResponsesAPIConfig from litellm.types.llms.openai import ( ResponsesAPIOptionalRequestParams, ResponsesAPIResponse, - ResponsesAPIStreamingResponse, ) from litellm.types.router import GenericLiteLLMParams from litellm.types.utils import LlmProviders @@ -42,6 +42,7 @@ if TYPE_CHECKING: # to resume and requests are always executed inline. _STATEFUL_PARAMS: Final = ("previous_response_id", "background") _CANCEL_RESPONSE_ADAPTER: Final = TypeAdapter(dict[str, object]) +_BODY_FRAMING_HEADERS: Final = frozenset({"content-encoding", "content-length"}) class ApodexResponsesConfig(OpenAIResponsesAPIConfig): @@ -80,10 +81,22 @@ class ApodexResponsesConfig(OpenAIResponsesAPIConfig): raw_response: httpx.Response, logging_obj: LiteLLMLoggingObj, ) -> ResponsesAPIResponse: - payload: Final = _CANCEL_RESPONSE_ADAPTER.validate_json(raw_response.content) + """Backfill the fields Apodex omits from a cancel payload but ResponsesAPIResponse requires.""" + try: + payload: Final = _CANCEL_RESPONSE_ADAPTER.validate_json(raw_response.content) + except ValidationError: + return super().transform_cancel_response_api_response( + raw_response=raw_response, + logging_obj=logging_obj, + ) + normalized_response: Final = httpx.Response( status_code=raw_response.status_code, - headers=raw_response.headers, + # Content-Encoding and Content-Length describe the body being replaced here; + # carrying them over makes httpx try to decompress plain JSON on read. + headers={ + name: value for name, value in raw_response.headers.items() if name.lower() not in _BODY_FRAMING_HEADERS + }, json={ **payload, "created_at": payload.get("created_at", int(time())), @@ -95,40 +108,6 @@ class ApodexResponsesConfig(OpenAIResponsesAPIConfig): logging_obj=logging_obj, ) - def transform_streaming_response( - self, - model: str, - parsed_chunk: dict, # mutable-ok: matches the base-class signature - logging_obj: LiteLLMLoggingObj, - ) -> ResponsesAPIStreamingResponse: - swarm: Final = parsed_chunk.get("swarm") - swarm_data: Final = swarm.get("data") if isinstance(swarm, dict) else None - if ( - parsed_chunk.get("type") != "response.swarm.llm_delta" - or not isinstance(swarm_data, dict) - or swarm_data.get("channel") != "output_text" - or not isinstance(swarm_data.get("delta"), str) - ): - return super().transform_streaming_response( - model=model, - parsed_chunk=parsed_chunk, - logging_obj=logging_obj, - ) - - response_id: Final = str(parsed_chunk.get("response_id", "")) - return super().transform_streaming_response( - model=model, - parsed_chunk={ - "type": "response.output_text.delta", - "item_id": f"msg_{response_id}", - "output_index": 0, - "content_index": 0, - "delta": swarm_data["delta"], - "sequence_number": parsed_chunk.get("sequence_number", 0), - }, - logging_obj=logging_obj, - ) - def get_supported_openai_params(self, model: str) -> list: # mutable-ok: matches the base-class signature inherited: Final = super().get_supported_openai_params(model) if is_deep_research_model(model): diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 6e9a4c90130..250477bd81d 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -48297,9 +48297,9 @@ "supports_audio_output": true }, "apodex/apodex-1.1": { - "max_tokens": 262144, + "max_tokens": 65536, "max_input_tokens": 262144, - "max_output_tokens": 262144, + "max_output_tokens": 65536, "input_cost_per_token": 3e-07, "cache_read_input_token_cost": 3e-08, "output_cost_per_token": 3e-06, @@ -48318,14 +48318,15 @@ "supports_native_streaming": true, "supports_prompt_caching": true, "supports_reasoning": true, + "supports_response_schema": false, "supports_system_messages": true, "supports_tool_choice": true, "supports_vision": false }, "apodex/apodex-1.1-mini": { - "max_tokens": 262144, + "max_tokens": 65536, "max_input_tokens": 262144, - "max_output_tokens": 262144, + "max_output_tokens": 65536, "input_cost_per_token": 1e-07, "cache_read_input_token_cost": 1e-08, "output_cost_per_token": 1e-06, @@ -48344,6 +48345,7 @@ "supports_native_streaming": true, "supports_prompt_caching": true, "supports_reasoning": true, + "supports_response_schema": false, "supports_system_messages": true, "supports_tool_choice": true, "supports_vision": false @@ -48404,7 +48406,6 @@ "mode": "chat", "source": "https://platform.apodex.ai/docs/pricing", "supported_endpoints": [ - "/v1/chat/completions", "/v1/responses" ], "supports_function_calling": false, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 6e9a4c90130..250477bd81d 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -48297,9 +48297,9 @@ "supports_audio_output": true }, "apodex/apodex-1.1": { - "max_tokens": 262144, + "max_tokens": 65536, "max_input_tokens": 262144, - "max_output_tokens": 262144, + "max_output_tokens": 65536, "input_cost_per_token": 3e-07, "cache_read_input_token_cost": 3e-08, "output_cost_per_token": 3e-06, @@ -48318,14 +48318,15 @@ "supports_native_streaming": true, "supports_prompt_caching": true, "supports_reasoning": true, + "supports_response_schema": false, "supports_system_messages": true, "supports_tool_choice": true, "supports_vision": false }, "apodex/apodex-1.1-mini": { - "max_tokens": 262144, + "max_tokens": 65536, "max_input_tokens": 262144, - "max_output_tokens": 262144, + "max_output_tokens": 65536, "input_cost_per_token": 1e-07, "cache_read_input_token_cost": 1e-08, "output_cost_per_token": 1e-06, @@ -48344,6 +48345,7 @@ "supports_native_streaming": true, "supports_prompt_caching": true, "supports_reasoning": true, + "supports_response_schema": false, "supports_system_messages": true, "supports_tool_choice": true, "supports_vision": false @@ -48404,7 +48406,6 @@ "mode": "chat", "source": "https://platform.apodex.ai/docs/pricing", "supported_endpoints": [ - "/v1/chat/completions", "/v1/responses" ], "supports_function_calling": false, diff --git a/tests/test_litellm/llms/apodex/test_apodex_chat_transformation.py b/tests/test_litellm/llms/apodex/test_apodex_chat_transformation.py index 61fccf2483a..da31464d5cb 100644 --- a/tests/test_litellm/llms/apodex/test_apodex_chat_transformation.py +++ b/tests/test_litellm/llms/apodex/test_apodex_chat_transformation.py @@ -99,10 +99,11 @@ class TestProviderResolution: class TestStreamDefault: - """Apodex defaults `stream` to true, so a non-streaming call has to pin it to false. + """The Deep Research tiers default `stream` to true, so a non-streaming call must pin it false. - Regression guard: the OpenAI SDK drops `stream` from the body when it is false, - which would leave Apodex streaming SSE at a call that cannot parse it. + Regression guard: the OpenAI chat handler pops `stream` out of the params it + forwards, which would leave those tiers streaming SSE at a call that cannot + parse it. The core models default to false but are pinned the same way. """ def test_non_streaming_call_pins_stream_false(self): diff --git a/tests/test_litellm/llms/apodex/test_apodex_common_utils.py b/tests/test_litellm/llms/apodex/test_apodex_common_utils.py index bbbe1c2f052..2da109c6ec4 100644 --- a/tests/test_litellm/llms/apodex/test_apodex_common_utils.py +++ b/tests/test_litellm/llms/apodex/test_apodex_common_utils.py @@ -99,6 +99,9 @@ class TestModelMetadata: assert info["litellm_provider"] == "apodex" assert info["mode"] == "chat" assert info["max_input_tokens"] == 262144 + # GET /v1/models reports max_completion_tokens 65536, well under the context window + assert info["max_output_tokens"] == 65536 + assert info["max_tokens"] == info["max_output_tokens"] assert info["input_cost_per_token"] == 3e-07 assert info["cache_read_input_token_cost"] == 3e-08 assert info["output_cost_per_token"] == 3e-06 @@ -126,6 +129,17 @@ class TestModelMetadata: """Apodex serves /v1/messages for the core models only.""" assert "/v1/messages" not in model_cost[f"apodex/{model}"]["supported_endpoints"] + def test_discover_is_responses_only(self, model_cost: dict): + """The Discover tiers answer 400 unsupported_api on /v1/chat/completions.""" + assert model_cost["apodex/apodex-1-1-deep-discover"]["supported_endpoints"] == ["/v1/responses"] + + @pytest.mark.parametrize("model", ("apodex-1-1-deep-research", "apodex-1-1-deep-solve")) + def test_the_other_deep_tiers_keep_chat_completions(self, model_cost: dict, model: str): + assert model_cost[f"apodex/{model}"]["supported_endpoints"] == [ + "/v1/chat/completions", + "/v1/responses", + ] + def test_backup_cost_map_in_sync(self, model_cost: dict): with open(REPO_ROOT / "litellm" / "model_prices_and_context_window_backup.json") as f: backup = json.load(f) diff --git a/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py b/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py index 44eb0cb4b1e..ac8e6db7cc4 100644 --- a/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py +++ b/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py @@ -6,6 +6,8 @@ Research tiers keep server-side state, so the parameter contract is keyed off the model rather than applied provider-wide. """ +import gzip + import httpx import pytest @@ -113,9 +115,7 @@ class TestStreamDefault: raise RuntimeError("captured") with pytest.raises(Exception, match="captured"): - await litellm.aresponses( - model=DEEP_RESEARCH_MODEL, input="hi", stream=True, client=CapturingHandler() - ) + await litellm.aresponses(model=DEEP_RESEARCH_MODEL, input="hi", stream=True, client=CapturingHandler()) assert captured["body"]["stream"] is True @@ -203,21 +203,39 @@ class TestDeepResearchKeepsState: assert response.output == [] assert response.created_at > 0 - def test_deep_research_output_delta_is_normalized(self): - config = _responses_config("apodex-1-1-deep-research") - event = config.transform_streaming_response( - model="apodex-1-1-deep-research", - parsed_chunk={ - "type": "response.swarm.llm_delta", - "response_id": "w_123", - "sequence_number": 12, - "swarm": { - "agent_id": "reporter", - "data": {"channel": "output_text", "delta": "final answer"}, - }, + def test_cancel_response_survives_a_compressed_upstream_response(self): + """The body is rebuilt, so the original framing headers must not follow it. + + httpx decodes on read, so carrying Content-Encoding over from the compressed + upstream response makes it try to gunzip the plain JSON replacement. + """ + body = b'{"id": "resp_1", "object": "response", "status": "cancelled"}' + # As it arrives off the wire: httpx decodes the body but leaves the header in place + upstream = httpx.Response( + 200, + headers={ + "content-encoding": "gzip", + "x-ratelimit-remaining-requests": "42", }, + content=gzip.compress(body), + ) + assert upstream.content == body + + config = _responses_config("apodex-1-1-deep-research") + response = config.transform_cancel_response_api_response( + raw_response=upstream, logging_obj=None, ) - assert event.type == "response.output_text.delta" - assert event.item_id == "msg_w_123" - assert event.delta == "final answer" + + assert response.id == "resp_1" + assert response.status == "cancelled" + assert response._hidden_params["headers"]["x-ratelimit-remaining-requests"] == "42" + + def test_non_json_cancel_body_raises_the_provider_error(self): + """The gateway answers a timed-out cancel with an HTML 504, not the JSON envelope.""" + config = _responses_config("apodex-1-1-deep-research") + with pytest.raises(Exception, match="gateway timeout"): + config.transform_cancel_response_api_response( + raw_response=httpx.Response(504, content=b"gateway timeout"), + logging_obj=None, + )