fix(apodex): correct model metadata and the cancel-response rebuild

Cross-checked the provider against platform.apodex.ai/docs and a live
GET /v1/models call.

- apodex-1.1 and apodex-1.1-mini advertised 256K max output; /v1/models
  reports 65536, and max_tokens is the legacy alias of max_output_tokens
- apodex-1-1-deep-discover is Responses-API-only; /v1/chat/completions
  answers 400 unsupported_api for the Discover tiers
- core models do not support response_format, so state it explicitly
- transform_cancel_response_api_response carried Content-Encoding over to
  a response whose body it had already replaced, so httpx tried to
  decompress plain JSON on read. A non-JSON body (the gateway answers a
  timed-out cancel with an HTML 504) also escaped as a pydantic
  ValidationError instead of the provider error
- drop the undocumented response.swarm.llm_delta mapping
- only the Deep Research tiers default stream to true; the core models
  follow OpenAI and default it to false
This commit is contained in:
zhanghanduo 2026-08-16 17:00:22 +08:00
parent 4ace5c8db3
commit c3f8e5aa67
9 changed files with 94 additions and 77 deletions

View file

@ -351,7 +351,6 @@ def get_llm_provider(
dynamic_api_key = get_secret_str("META_API_KEY")
elif endpoint == litellm.ApodexChatConfig.API_BASE_URL:
custom_llm_provider = "apodex" # rebind-ok: dispatch chain resolves in place
# rebind-ok: dispatch chain resolves in place
dynamic_api_key = litellm.ApodexChatConfig.get_api_key()
if api_base is not None and not isinstance(api_base, str):

View file

@ -529,7 +529,8 @@ def anthropic_messages_handler(
anthropic_messages_provider_config = OpenAILikeAnthropicMessagesConfig()
if anthropic_messages_provider_config is None:
# Route to Responses API for OpenAI / Azure, chat/completions for everything else.
# Route to a Responses API for the providers that serve one, chat/completions
# for everything else.
_shared_kwargs: Final = dict(
max_tokens=max_tokens,
messages=messages,

View file

@ -1,12 +1,14 @@
"""
Apodex chat completions — OpenAI-compatible, with two provider quirks:
- `stream` defaults to true upstream, so a non-streaming call has to say so
explicitly or Apodex answers with SSE that a plain call cannot parse
- the Deep Research tiers default `stream` to true, so a non-streaming call has
to say so explicitly or Apodex answers with SSE that a plain call cannot
parse. The core models follow OpenAI and default it to false
- the Deep Research tiers ignore sampling parameters and reject OpenAI-style
tools; only the core models take them
Ref: https://platform.apodex.ai/docs/chat-completions
https://platform.apodex.ai/docs/models
"""
from collections.abc import Mapping
@ -104,9 +106,10 @@ class ApodexChatConfig(OpenAIGPTConfig):
if renamed.get("stream"):
return renamed
# The OpenAI SDK drops `stream` from the body when it is false, which would
# leave Apodex on its streaming default. extra_body is merged into the
# request body by the SDK, so it survives that drop.
# The OpenAI chat handler pops `stream` out of the params it forwards
# (litellm/llms/openai/openai.py), which would leave the Deep Research tiers
# on their streaming default. extra_body is merged into the request body
# further down, so it survives that pop.
requested_extra_body: Final = renamed.get("extra_body")
extra_body: Final = (
requested_extra_body if isinstance(requested_extra_body, Mapping) else {} # mutable-ok: JSON request body

View file

@ -8,7 +8,8 @@ provider-wide:
- core models are a stateless subset: `store` is forced to false, and
`previous_response_id` or `background` come back as HTTP 400
- the Deep Research tiers keep server-side state, so they take all three
- both default `stream` to true, so a non-streaming call has to say so
- the Deep Research tiers default `stream` to true, so a non-streaming call has
to say so; pinning it for the core models too keeps one code path
Ref: https://platform.apodex.ai/docs/responses-api
https://platform.apodex.ai/docs/models
@ -21,14 +22,13 @@ from time import time
from typing import TYPE_CHECKING, Final
import httpx
from pydantic import TypeAdapter
from pydantic import TypeAdapter, ValidationError
import litellm
from litellm.llms.openai.responses.transformation import OpenAIResponsesAPIConfig
from litellm.types.llms.openai import (
ResponsesAPIOptionalRequestParams,
ResponsesAPIResponse,
ResponsesAPIStreamingResponse,
)
from litellm.types.router import GenericLiteLLMParams
from litellm.types.utils import LlmProviders
@ -42,6 +42,7 @@ if TYPE_CHECKING:
# to resume and requests are always executed inline.
_STATEFUL_PARAMS: Final = ("previous_response_id", "background")
_CANCEL_RESPONSE_ADAPTER: Final = TypeAdapter(dict[str, object])
_BODY_FRAMING_HEADERS: Final = frozenset({"content-encoding", "content-length"})
class ApodexResponsesConfig(OpenAIResponsesAPIConfig):
@ -80,10 +81,22 @@ class ApodexResponsesConfig(OpenAIResponsesAPIConfig):
raw_response: httpx.Response,
logging_obj: LiteLLMLoggingObj,
) -> ResponsesAPIResponse:
payload: Final = _CANCEL_RESPONSE_ADAPTER.validate_json(raw_response.content)
"""Backfill the fields Apodex omits from a cancel payload but ResponsesAPIResponse requires."""
try:
payload: Final = _CANCEL_RESPONSE_ADAPTER.validate_json(raw_response.content)
except ValidationError:
return super().transform_cancel_response_api_response(
raw_response=raw_response,
logging_obj=logging_obj,
)
normalized_response: Final = httpx.Response(
status_code=raw_response.status_code,
headers=raw_response.headers,
# Content-Encoding and Content-Length describe the body being replaced here;
# carrying them over makes httpx try to decompress plain JSON on read.
headers={
name: value for name, value in raw_response.headers.items() if name.lower() not in _BODY_FRAMING_HEADERS
},
json={
**payload,
"created_at": payload.get("created_at", int(time())),
@ -95,40 +108,6 @@ class ApodexResponsesConfig(OpenAIResponsesAPIConfig):
logging_obj=logging_obj,
)
def transform_streaming_response(
self,
model: str,
parsed_chunk: dict, # mutable-ok: matches the base-class signature
logging_obj: LiteLLMLoggingObj,
) -> ResponsesAPIStreamingResponse:
swarm: Final = parsed_chunk.get("swarm")
swarm_data: Final = swarm.get("data") if isinstance(swarm, dict) else None
if (
parsed_chunk.get("type") != "response.swarm.llm_delta"
or not isinstance(swarm_data, dict)
or swarm_data.get("channel") != "output_text"
or not isinstance(swarm_data.get("delta"), str)
):
return super().transform_streaming_response(
model=model,
parsed_chunk=parsed_chunk,
logging_obj=logging_obj,
)
response_id: Final = str(parsed_chunk.get("response_id", ""))
return super().transform_streaming_response(
model=model,
parsed_chunk={
"type": "response.output_text.delta",
"item_id": f"msg_{response_id}",
"output_index": 0,
"content_index": 0,
"delta": swarm_data["delta"],
"sequence_number": parsed_chunk.get("sequence_number", 0),
},
logging_obj=logging_obj,
)
def get_supported_openai_params(self, model: str) -> list: # mutable-ok: matches the base-class signature
inherited: Final = super().get_supported_openai_params(model)
if is_deep_research_model(model):

View file

@ -48297,9 +48297,9 @@
"supports_audio_output": true
},
"apodex/apodex-1.1": {
"max_tokens": 262144,
"max_tokens": 65536,
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"max_output_tokens": 65536,
"input_cost_per_token": 3e-07,
"cache_read_input_token_cost": 3e-08,
"output_cost_per_token": 3e-06,
@ -48318,14 +48318,15 @@
"supports_native_streaming": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": false,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": false
},
"apodex/apodex-1.1-mini": {
"max_tokens": 262144,
"max_tokens": 65536,
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"max_output_tokens": 65536,
"input_cost_per_token": 1e-07,
"cache_read_input_token_cost": 1e-08,
"output_cost_per_token": 1e-06,
@ -48344,6 +48345,7 @@
"supports_native_streaming": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": false,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": false
@ -48404,7 +48406,6 @@
"mode": "chat",
"source": "https://platform.apodex.ai/docs/pricing",
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
],
"supports_function_calling": false,

View file

@ -48297,9 +48297,9 @@
"supports_audio_output": true
},
"apodex/apodex-1.1": {
"max_tokens": 262144,
"max_tokens": 65536,
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"max_output_tokens": 65536,
"input_cost_per_token": 3e-07,
"cache_read_input_token_cost": 3e-08,
"output_cost_per_token": 3e-06,
@ -48318,14 +48318,15 @@
"supports_native_streaming": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": false,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": false
},
"apodex/apodex-1.1-mini": {
"max_tokens": 262144,
"max_tokens": 65536,
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"max_output_tokens": 65536,
"input_cost_per_token": 1e-07,
"cache_read_input_token_cost": 1e-08,
"output_cost_per_token": 1e-06,
@ -48344,6 +48345,7 @@
"supports_native_streaming": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": false,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": false
@ -48404,7 +48406,6 @@
"mode": "chat",
"source": "https://platform.apodex.ai/docs/pricing",
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
],
"supports_function_calling": false,

View file

@ -99,10 +99,11 @@ class TestProviderResolution:
class TestStreamDefault:
"""Apodex defaults `stream` to true, so a non-streaming call has to pin it to false.
"""The Deep Research tiers default `stream` to true, so a non-streaming call must pin it false.
Regression guard: the OpenAI SDK drops `stream` from the body when it is false,
which would leave Apodex streaming SSE at a call that cannot parse it.
Regression guard: the OpenAI chat handler pops `stream` out of the params it
forwards, which would leave those tiers streaming SSE at a call that cannot
parse it. The core models default to false but are pinned the same way.
"""
def test_non_streaming_call_pins_stream_false(self):

View file

@ -99,6 +99,9 @@ class TestModelMetadata:
assert info["litellm_provider"] == "apodex"
assert info["mode"] == "chat"
assert info["max_input_tokens"] == 262144
# GET /v1/models reports max_completion_tokens 65536, well under the context window
assert info["max_output_tokens"] == 65536
assert info["max_tokens"] == info["max_output_tokens"]
assert info["input_cost_per_token"] == 3e-07
assert info["cache_read_input_token_cost"] == 3e-08
assert info["output_cost_per_token"] == 3e-06
@ -126,6 +129,17 @@ class TestModelMetadata:
"""Apodex serves /v1/messages for the core models only."""
assert "/v1/messages" not in model_cost[f"apodex/{model}"]["supported_endpoints"]
def test_discover_is_responses_only(self, model_cost: dict):
"""The Discover tiers answer 400 unsupported_api on /v1/chat/completions."""
assert model_cost["apodex/apodex-1-1-deep-discover"]["supported_endpoints"] == ["/v1/responses"]
@pytest.mark.parametrize("model", ("apodex-1-1-deep-research", "apodex-1-1-deep-solve"))
def test_the_other_deep_tiers_keep_chat_completions(self, model_cost: dict, model: str):
assert model_cost[f"apodex/{model}"]["supported_endpoints"] == [
"/v1/chat/completions",
"/v1/responses",
]
def test_backup_cost_map_in_sync(self, model_cost: dict):
with open(REPO_ROOT / "litellm" / "model_prices_and_context_window_backup.json") as f:
backup = json.load(f)

View file

@ -6,6 +6,8 @@ Research tiers keep server-side state, so the parameter contract is keyed off
the model rather than applied provider-wide.
"""
import gzip
import httpx
import pytest
@ -113,9 +115,7 @@ class TestStreamDefault:
raise RuntimeError("captured")
with pytest.raises(Exception, match="captured"):
await litellm.aresponses(
model=DEEP_RESEARCH_MODEL, input="hi", stream=True, client=CapturingHandler()
)
await litellm.aresponses(model=DEEP_RESEARCH_MODEL, input="hi", stream=True, client=CapturingHandler())
assert captured["body"]["stream"] is True
@ -203,21 +203,39 @@ class TestDeepResearchKeepsState:
assert response.output == []
assert response.created_at > 0
def test_deep_research_output_delta_is_normalized(self):
config = _responses_config("apodex-1-1-deep-research")
event = config.transform_streaming_response(
model="apodex-1-1-deep-research",
parsed_chunk={
"type": "response.swarm.llm_delta",
"response_id": "w_123",
"sequence_number": 12,
"swarm": {
"agent_id": "reporter",
"data": {"channel": "output_text", "delta": "final answer"},
},
def test_cancel_response_survives_a_compressed_upstream_response(self):
"""The body is rebuilt, so the original framing headers must not follow it.
httpx decodes on read, so carrying Content-Encoding over from the compressed
upstream response makes it try to gunzip the plain JSON replacement.
"""
body = b'{"id": "resp_1", "object": "response", "status": "cancelled"}'
# As it arrives off the wire: httpx decodes the body but leaves the header in place
upstream = httpx.Response(
200,
headers={
"content-encoding": "gzip",
"x-ratelimit-remaining-requests": "42",
},
content=gzip.compress(body),
)
assert upstream.content == body
config = _responses_config("apodex-1-1-deep-research")
response = config.transform_cancel_response_api_response(
raw_response=upstream,
logging_obj=None,
)
assert event.type == "response.output_text.delta"
assert event.item_id == "msg_w_123"
assert event.delta == "final answer"
assert response.id == "resp_1"
assert response.status == "cancelled"
assert response._hidden_params["headers"]["x-ratelimit-remaining-requests"] == "42"
def test_non_json_cancel_body_raises_the_provider_error(self):
"""The gateway answers a timed-out cancel with an HTML 504, not the JSON envelope."""
config = _responses_config("apodex-1-1-deep-research")
with pytest.raises(Exception, match="gateway timeout"):
config.transform_cancel_response_api_response(
raw_response=httpx.Response(504, content=b"<html>gateway timeout</html>"),
logging_obj=None,
)