This commit is contained in:
Hans 2026-08-26 10:12:48 -07:00 • committed by GitHub
commit a23c6394a2
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
22 changed files with 1931 additions and 38 deletions

View file

@ -279,6 +279,7 @@ curl -X POST 'http://0.0.0.0:4000/v1/chat/completions' \
| [Anthropic (`anthropic`)](https://docs.litellm.ai/docs/providers/anthropic) | ✅ | ✅ | ✅ | | | | | | ✅ | |
| [Anthropic Text (`anthropic_text`)](https://docs.litellm.ai/docs/providers/anthropic) | ✅ | ✅ | ✅ | | | | | | ✅ | |
| [Anyscale](https://docs.litellm.ai/docs/providers/anyscale) | ✅ | ✅ | ✅ | | | | | | | |
| [Apodex (`apodex`)](https://docs.litellm.ai/docs/providers/apodex) | ✅ | ✅ | ✅ | | | | | | | |
| [AssemblyAI (`assemblyai`)](https://docs.litellm.ai/docs/pass_through/assembly_ai) | ✅ | ✅ | ✅ | | | ✅ | | | | |
| [Auto Router (`auto_router`)](https://docs.litellm.ai/docs/proxy/auto_routing) | ✅ | ✅ | ✅ | | | | | | | |
| [AWS - Bedrock (`bedrock`)](https://docs.litellm.ai/docs/providers/bedrock) | ✅ | ✅ | ✅ | ✅ | | | | | | ✅ |

View file

@ -649,6 +649,7 @@ snowflake_models: Set = set()
gradient_ai_models: Set = set()
llama_models: Set = set()
nscale_models: Set = set()
apodex_models: Set = set() # mutable-ok: provider model registry is populated during initialization
nebius_models: Set = set()
nebius_embedding_models: Set = set()
aiml_models: Set = set()
@ -841,6 +842,8 @@ def _populate_provider_model_sets(model_cost_map: Dict) -> None:
llama_models.add(key)
elif value.get("litellm_provider") == "nscale":
nscale_models.add(key)
elif value.get("litellm_provider") == "apodex":
apodex_models.add(key)
elif value.get("litellm_provider") == "azure_ai":
azure_ai_models.add(key)
elif value.get("litellm_provider") == "voyage":
@ -1065,6 +1068,7 @@ model_list = list(
| llama_models
| featherless_ai_models
| nscale_models
| apodex_models
| deepgram_models
| elevenlabs_models
| dashscope_models
@ -1169,6 +1173,7 @@ def _build_models_by_provider() -> dict:
"gradient_ai": gradient_ai_models,
"meta_llama": llama_models,
"nscale": nscale_models,
"apodex": apodex_models,
"featherless_ai": featherless_ai_models,
"deepgram": deepgram_models,
"elevenlabs": elevenlabs_models,
@ -1798,6 +1803,9 @@ if TYPE_CHECKING:
from .llms.perplexity.responses.transformation import (
PerplexityResponsesConfig as PerplexityResponsesConfig,
)
from .llms.apodex.responses.transformation import (
ApodexResponsesConfig as ApodexResponsesConfig,
)
from .llms.databricks.responses.transformation import (
DatabricksResponsesAPIConfig as DatabricksResponsesAPIConfig,
)
@ -1874,6 +1882,7 @@ if TYPE_CHECKING:
PerplexityChatConfig as _PerplexityChatConfig,
)
from .llms.nscale.chat.transformation import NscaleConfig as _NscaleConfig
from .llms.apodex.chat.transformation import ApodexChatConfig as _ApodexChatConfig
from .llms.watsonx.chat.transformation import (
IBMWatsonXChatConfig as _IBMWatsonXChatConfig,
)
@ -1909,6 +1918,7 @@ if TYPE_CHECKING:
AzureOpenAIO1Config: Type[_AzureOpenAIO1Config]
PerplexityChatConfig: Type[_PerplexityChatConfig]
NscaleConfig: Type[_NscaleConfig]
ApodexChatConfig: Type[_ApodexChatConfig]
IBMWatsonXChatConfig: Type[_IBMWatsonXChatConfig]
IBMWatsonXAIConfig: Type[_IBMWatsonXAIConfig]
LiteLLMProxyChatConfig: Type[_LiteLLMProxyChatConfig]

View file

@ -239,6 +239,7 @@ LLM_CONFIG_NAMES: Final = (
"HostedVLLMResponsesAPIConfig",
"VolcEngineResponsesAPIConfig",
"PerplexityResponsesConfig",
"ApodexResponsesConfig",
"DatabricksResponsesAPIConfig",
"OpenRouterResponsesAPIConfig",
"BedrockMantleResponsesAPIConfig",
@ -293,6 +294,7 @@ LLM_CONFIG_NAMES: Final = (
"LmStudioEmbeddingConfig",
"NscaleConfig",
"PerplexityChatConfig",
"ApodexChatConfig",
"AzureOpenAIO1Config",
"IBMWatsonXAIConfig",
"IBMWatsonXChatConfig",
@ -967,6 +969,10 @@ _LLM_CONFIGS_IMPORT_MAP: Final = {
".llms.perplexity.responses.transformation",
"PerplexityResponsesConfig",
),
"ApodexResponsesConfig": (
".llms.apodex.responses.transformation",
"ApodexResponsesConfig",
),
"DatabricksResponsesAPIConfig": (
".llms.databricks.responses.transformation",
"DatabricksResponsesAPIConfig",
@ -1120,6 +1126,7 @@ _LLM_CONFIGS_IMPORT_MAP: Final = {
".llms.perplexity.chat.transformation",
"PerplexityChatConfig",
),
"ApodexChatConfig": (".llms.apodex.chat.transformation", "ApodexChatConfig"),
"AzureOpenAIO1Config": (
".llms.azure.chat.o_series_transformation",
"AzureOpenAIO1Config",

View file

@ -789,6 +789,7 @@ openai_compatible_endpoints: Final[list] = [
"https://api.libertai.io/v1",
"https://pinstripes.io/v1",
"https://api.meta.ai/v1",
"https://api.apodex.ai/v1",
"https://api.cognition.ai/v1",
"https://api.scx.ai/v1",
]
@ -858,6 +859,7 @@ openai_compatible_providers: Final[list] = [
"pinstripes", # Pinstripes - JSON-configured provider
"darkbloom",
"meta", # Meta Model API (Muse Spark) - JSON-configured provider
"apodex",
"cognition",
"scx-ai",
]

View file

@ -357,6 +357,9 @@ def get_llm_provider(
elif endpoint == "https://api.meta.ai/v1":
custom_llm_provider = "meta"
dynamic_api_key = get_secret_str("META_API_KEY")
elif endpoint == litellm.ApodexChatConfig.API_BASE_URL:
custom_llm_provider = "apodex" # rebind-ok: dispatch chain resolves in place
dynamic_api_key = litellm.ApodexChatConfig.get_api_key()
elif (json_provider := JSONProviderRegistry.get_by_base_url(endpoint)) is not None:
custom_llm_provider = json_provider.slug
dynamic_api_key = api_key if api_key is not None else get_secret_str(json_provider.api_key_env)
@ -765,6 +768,9 @@ def _get_openai_compatible_provider_info(
api_base,
dynamic_api_key,
) = litellm.NscaleConfig()._get_openai_compatible_provider_info(api_base=api_base, api_key=api_key)
elif custom_llm_provider == "apodex":
api_base = litellm.ApodexChatConfig.get_api_base(api_base) # rebind-ok: dispatch chain resolves in place
dynamic_api_key = litellm.ApodexChatConfig.get_api_key(api_key) # rebind-ok: resolved in place
elif custom_llm_provider == "heroku":
(
api_base,

View file

@ -39,9 +39,9 @@ from ..utils import is_reasoning_auto_summary_enabled
from .interceptors import get_messages_interceptors
from .utils import AnthropicMessagesRequestUtils, mock_response
# Providers that are routed directly to the OpenAI Responses API instead of
# Providers that are routed directly to a Responses API instead of
# going through chat/completions.
_RESPONSES_API_PROVIDERS: Final = frozenset({"openai"})
_RESPONSES_API_PROVIDERS: Final = frozenset({"apodex", "openai"})
def _bridges_to_responses_api(model: str, custom_llm_provider: str) -> bool:
@ -567,11 +567,13 @@ def anthropic_messages_handler(
anthropic_messages_provider_config = OpenAILikeAnthropicMessagesConfig()
if anthropic_messages_provider_config is None:
# Route to Responses API for OpenAI / Azure, chat/completions for everything else.
# Route to a Responses API for the providers that serve one, chat/completions
# for everything else.
downstream_model: Final = model if custom_llm_provider == "apodex" else original_model
_shared_kwargs: Final = dict(
max_tokens=max_tokens,
messages=messages,
model=original_model,
model=downstream_model,
metadata=metadata,
stop_sequences=stop_sequences,
stream=stream,

View file

@ -36,6 +36,8 @@ class AnthropicResponsesStreamWrapper:
self.model = model
self._message_id: str = f"msg_{uuid.uuid4()}"
self._current_block_index: int = -1
self._open_block_index: int | None = None
self._open_block_type: str | None = None
# Map item_id -> content_block_index so we can stop the right block later
self._item_id_to_block_index: dict[str, int] = {}
# Track open function_call items by item_id so we can emit tool_use start
@ -68,17 +70,49 @@ class AnthropicResponsesStreamWrapper:
self._current_block_index += 1
return self._current_block_index
def _open_block(self, item_id: str | None, content_block: Mapping[str, Any]) -> int:
block_idx = self._next_block_index()
if item_id:
self._item_id_to_block_index[item_id] = block_idx
def _close_open_block(self) -> None:
if self._open_block_index is None:
return
self._chunk_queue.append(
{ # mutable-ok: queued Anthropic event payload
"type": "content_block_stop",
"index": self._open_block_index,
}
)
self._open_block_index = None
self._open_block_type = None
def _start_block(self, block_idx: int, block_type: str, content_block: Mapping[str, object]) -> None:
self._close_open_block()
self._chunk_queue.append(
{
"type": "content_block_start",
"index": block_idx,
"content_block": content_block,
"content_block": dict(content_block), # mutable-ok: queued Anthropic event payload
}
)
self._open_block_index = block_idx
self._open_block_type = block_type
def _get_or_start_block(
self,
item_id: str | None,
block_type: str,
content_block: Mapping[str, object],
) -> int:
mapped_index: Final = self._item_id_to_block_index.get(item_id) if item_id else None
if mapped_index is not None and mapped_index == self._open_block_index:
return mapped_index
# A resumed item whose block already closed needs a fresh one: Anthropic rejects
# a delta addressed to a stopped block. Providers that reuse one item id for a
# whole run, then interleave channels, land here.
if item_id is None and self._open_block_index is not None and self._open_block_type == block_type:
return self._open_block_index
block_idx: Final = self._next_block_index()
if item_id:
self._item_id_to_block_index[item_id] = block_idx
self._start_block(block_idx, block_type, content_block)
return block_idx
def _process_event(self, event: Any) -> None:
@ -106,16 +140,26 @@ class AnthropicResponsesStreamWrapper:
item_id = getattr(item, "id", None) or (item.get("id") if isinstance(item, dict) else None)
if item_type == "message":
self._open_block(item_id, {"type": "text", "text": ""})
block_idx = self._next_block_index()
if item_id:
self._item_id_to_block_index[item_id] = block_idx
self._start_block(
block_idx,
"text",
{"type": "text", "text": ""}, # mutable-ok: Anthropic content block payload
)
elif item_type == "function_call":
call_id: Final = (
getattr(item, "call_id", None) or (item.get("call_id") if isinstance(item, dict) else None) or ""
)
name = getattr(item, "name", None) or (item.get("name") if isinstance(item, dict) else None) or ""
block_idx = self._next_block_index()
if item_id:
self._item_id_to_block_index[item_id] = block_idx
self._pending_tool_ids[item_id] = call_id
self._open_block(
item_id,
self._start_block(
block_idx,
"tool_use",
{
"type": "tool_use",
"id": call_id,
@ -129,16 +173,15 @@ class AnthropicResponsesStreamWrapper:
if event_type == "response.output_text.delta":
item_id = getattr(event, "item_id", None) or (event.get("item_id") if isinstance(event, dict) else None)
delta = getattr(event, "delta", "") or (event.get("delta", "") if isinstance(event, dict) else "")
block_idx = self._item_id_to_block_index.get(item_id, -1) if item_id else self._current_block_index
if block_idx < 0:
# Some providers (e.g. LMStudio) skip response.output_item.added,
# so no text block is open yet; synthesize content_block_start
# instead of emitting a delta with index -1
block_idx = self._open_block(item_id, {"type": "text", "text": ""})
text_block_idx: Final = self._get_or_start_block(
item_id=item_id,
block_type="text",
content_block={"type": "text", "text": ""}, # mutable-ok: Anthropic content block payload
)
self._chunk_queue.append(
{
"type": "content_block_delta",
"index": block_idx,
"index": text_block_idx,
"delta": {"type": "text_delta", "text": delta},
}
)
@ -148,18 +191,21 @@ class AnthropicResponsesStreamWrapper:
if event_type == "response.reasoning_summary_text.delta":
item_id = getattr(event, "item_id", None) or (event.get("item_id") if isinstance(event, dict) else None)
delta = getattr(event, "delta", "") or (event.get("delta", "") if isinstance(event, dict) else "")
block_idx = self._item_id_to_block_index.get(item_id, -1) if item_id else self._current_block_index
if block_idx < 0:
if not delta:
return
block_idx = self._open_block(
item_id,
{"type": "thinking", "thinking": "", "signature": ""}, # mutable-ok: API message payload
)
if not delta:
return
thinking_block_idx: Final = self._get_or_start_block(
item_id=item_id,
block_type="thinking",
content_block={ # mutable-ok: Anthropic content block payload
"type": "thinking",
"thinking": "",
"signature": "",
},
)
self._chunk_queue.append(
{
"type": "content_block_delta",
"index": block_idx,
"index": thinking_block_idx,
"delta": {"type": "thinking_delta", "thinking": delta},
}
)
@ -192,12 +238,8 @@ class AnthropicResponsesStreamWrapper:
block_idx = self._item_id_to_block_index.get(item_id, -1) if item_id else self._current_block_index
if block_idx < 0:
return
self._chunk_queue.append(
{
"type": "content_block_stop",
"index": block_idx,
}
)
if block_idx == self._open_block_index:
self._close_open_block()
return
# ---- response completed -> message_delta + message_stop ----
@ -206,6 +248,7 @@ class AnthropicResponsesStreamWrapper:
"response.failed",
"response.incomplete",
):
self._close_open_block()
response_obj: Final = getattr(event, "response", None) or (
event.get("response") if isinstance(event, dict) else None
)

View file

@ -0,0 +1,154 @@
"""
Apodex chat completions — OpenAI-compatible, with two provider quirks:
- the Deep Research tiers default `stream` to true, so a non-streaming call has
to say so explicitly or Apodex answers with SSE that a plain call cannot
parse. The core models follow OpenAI and default it to false
- the Deep Research tiers ignore sampling parameters and reject OpenAI-style
tools; only the core models take them
Ref: https://platform.apodex.ai/docs/chat-completions
https://platform.apodex.ai/docs/models
"""
from collections.abc import Mapping
from typing import Final
import litellm
from litellm.llms.openai.chat.gpt_transformation import OpenAIGPTConfig
from litellm.types.llms.openai import AllMessageValues
from ..common_utils import (
APODEX_API_BASE_URL,
get_apodex_api_base,
get_apodex_api_key,
is_deep_research_model,
is_responses_only_model,
)
_DEEP_RESEARCH_PARAMS: Final = (
"max_tokens",
"max_completion_tokens",
"stream",
"stream_options",
"extra_headers",
"max_retries",
)
_CORE_PARAMS: Final = (
*_DEEP_RESEARCH_PARAMS,
"temperature",
"top_p",
"stop",
"seed",
"n",
"tools",
"tool_choice",
"function_call",
"functions",
"parallel_tool_calls",
)
_PIN_NON_STREAMING: Final = "_apodex_pin_non_streaming"
class ApodexChatConfig(OpenAIGPTConfig):
"""
Reference: https://platform.apodex.ai/docs
API Key: APODEX_API_KEY
Default API Base: https://api.apodex.ai/v1
"""
API_BASE_URL = APODEX_API_BASE_URL
@property
def custom_llm_provider(self) -> str | None:
return "apodex"
@staticmethod
def get_api_key(api_key: str | None = None) -> str | None:
return get_apodex_api_key(api_key)
@staticmethod
def get_api_base(api_base: str | None = None) -> str | None:
return get_apodex_api_base(api_base)
def _get_openai_compatible_provider_info(
self, api_base: str | None, api_key: str | None
) -> tuple[str | None, str | None]:
return get_apodex_api_base(api_base), get_apodex_api_key(api_key)
def get_supported_openai_params(self, model: str) -> list: # mutable-ok: matches the base-class signature
supported: Final = _DEEP_RESEARCH_PARAMS if is_deep_research_model(model) else _CORE_PARAMS
return list(supported) # mutable-ok: matches the base-class signature
def map_openai_params(
self,
non_default_params: dict, # mutable-ok: matches the base-class signature
optional_params: dict, # mutable-ok: matches the base-class signature
model: str,
drop_params: bool,
) -> dict: # mutable-ok: matches the base-class signature
if is_responses_only_model(model):
raise litellm.BadRequestError(
message=f"apodex model {model} is only available through /v1/responses",
model=model,
llm_provider="apodex",
)
mapped: Final = super().map_openai_params(
non_default_params=non_default_params,
optional_params=optional_params,
model=model,
drop_params=drop_params,
)
renamed: Final = (
mapped
if "max_completion_tokens" not in mapped
else { # mutable-ok: JSON request body
**{ # mutable-ok: JSON request body
key: value for key, value in mapped.items() if key != "max_completion_tokens"
},
"max_tokens": mapped["max_completion_tokens"],
}
)
if renamed.get("stream"):
return renamed
return { # mutable-ok: JSON request body
**renamed,
_PIN_NON_STREAMING: True,
}
def transform_request(
self,
model: str,
messages: list[AllMessageValues], # mutable-ok: matches the base-class signature
optional_params: dict, # mutable-ok: matches the base-class signature
litellm_params: dict, # mutable-ok: matches the base-class signature
headers: dict, # mutable-ok: matches the base-class signature
) -> dict: # mutable-ok: JSON request body
pin_non_streaming: Final = bool(optional_params.get(_PIN_NON_STREAMING, False))
forwarded_params: Final = { # mutable-ok: base transformer requires a request dict
key: value for key, value in optional_params.items() if key != _PIN_NON_STREAMING
}
transformed: Final = super().transform_request(
model=model,
messages=messages,
optional_params=forwarded_params,
litellm_params=litellm_params,
headers=headers,
)
if not pin_non_streaming:
return transformed
requested_extra_body: Final = transformed.get("extra_body")
extra_body: Final = (
requested_extra_body if isinstance(requested_extra_body, Mapping) else {} # mutable-ok: JSON request body
)
return { # mutable-ok: JSON request body
**transformed,
"extra_body": {**extra_body, "stream": False}, # mutable-ok: JSON request body
}

View file

@ -0,0 +1,41 @@
"""
Shared helpers for the Apodex provider.
Apodex serves two model families on one base URL, and the model id picks which
contract applies. Core models (apodex-1.1, apodex-1.1-mini) are plain inference
with native sampling parameters. The Deep Research tiers run an agent that
plans, searches and iterates, so they ignore sampling parameters, reject
OpenAI-style tools, and keep server-side state.
Ref: https://platform.apodex.ai/docs/models
"""
from typing import Final
from litellm.secret_managers.main import get_secret_str
APODEX_API_BASE_URL: Final = "https://api.apodex.ai/v1"
_DEEP_RESEARCH_MARKER: Final = "-deep-"
_RESPONSES_ONLY_MODELS: Final = frozenset({"apodex-1-1-deep-discover"})
def strip_provider_prefix(model: str) -> str:
return model.rpartition("/")[2]
def is_deep_research_model(model: str) -> bool:
"""True for the Deep Research / Solve / Discover tiers, e.g. apodex-1-1-deep-solve."""
return _DEEP_RESEARCH_MARKER in strip_provider_prefix(model)
def is_responses_only_model(model: str) -> bool:
return strip_provider_prefix(model) in _RESPONSES_ONLY_MODELS
def get_apodex_api_key(api_key: str | None = None) -> str | None:
return api_key or get_secret_str("APODEX_API_KEY")
def get_apodex_api_base(api_base: str | None = None) -> str:
return api_base or get_secret_str("APODEX_API_BASE") or APODEX_API_BASE_URL

View file

@ -0,0 +1,52 @@
"""
Apodex Anthropic Messages — native passthrough for the core models only.
Apodex implements the Anthropic protocol itself at POST /v1/messages and serves
the core models there, so the payload is forwarded untranslated and
Anthropic-only features such as `thinking` and `cache_control` survive. The Deep
Research tiers are not served on that path, so `ProviderConfigManager` hands back
no config for them and they fall back to LiteLLM's Anthropic-to-chat-completions
translation.
Ref: https://platform.apodex.ai/docs/anthropic-messages
"""
from litellm.llms.openai_like.messages.transformation import (
OpenAILikeAnthropicMessagesConfig,
)
from ..common_utils import get_apodex_api_base, get_apodex_api_key
class ApodexAnthropicMessagesConfig(OpenAILikeAnthropicMessagesConfig):
@property
def custom_llm_provider(self) -> str | None:
return "apodex"
def should_strip_billing_metadata(self) -> bool:
return True
def validate_anthropic_messages_environment(
self,
headers: dict[str, str], # mutable-ok: matches the base-class signature
model: str,
messages: list[object], # mutable-ok: matches the base-class signature
optional_params: dict, # mutable-ok: matches the base-class signature
litellm_params: dict, # mutable-ok: matches the base-class signature
api_key: str | None = None,
api_base: str | None = None,
) -> tuple[dict[str, str], str | None]: # mutable-ok: matches the base-class signature
"""Fill in the Apodex credentials and base URL.
The returned api_base is what the handler hands to get_complete_url, so
resolving it here is enough to reach the native endpoint.
"""
return super().validate_anthropic_messages_environment(
headers=headers,
model=model,
messages=messages,
optional_params=optional_params,
litellm_params=litellm_params,
api_key=get_apodex_api_key(api_key),
api_base=get_apodex_api_base(api_base),
)

View file

@ -0,0 +1,230 @@
"""
Apodex Responses API — OpenAI-compatible, with a model-aware parameter contract.
Apodex serves /v1/responses for both model families but they accept different
subsets, so the restrictions here are keyed off the model rather than applied
provider-wide:
- core models are a stateless subset: `store` is forced to false, and
`previous_response_id` or `background` come back as HTTP 400
- the Deep Research tiers keep server-side state, so they take all three
- the Deep Research tiers default `stream` to true, so a non-streaming call has
to say so; pinning it for the core models too keeps one code path
Ref: https://platform.apodex.ai/docs/responses-api
https://platform.apodex.ai/docs/models
"""
from __future__ import annotations
from collections.abc import Mapping
from time import time
from types import MappingProxyType
from typing import TYPE_CHECKING, Final, NamedTuple
import httpx
from pydantic import TypeAdapter, ValidationError
import litellm
from litellm.llms.openai.responses.transformation import OpenAIResponsesAPIConfig
from litellm.types.llms.openai import (
ResponsesAPIOptionalRequestParams,
ResponsesAPIResponse,
ResponsesAPIStreamingResponse,
)
from litellm.types.router import GenericLiteLLMParams
from litellm.types.utils import LlmProviders
from ..common_utils import get_apodex_api_base, get_apodex_api_key, is_deep_research_model
if TYPE_CHECKING:
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
# Rejected by the core models with HTTP 400: there is no server-side conversation
# to resume and requests are always executed inline.
_STATEFUL_PARAMS: Final = ("previous_response_id", "background")
_CANCEL_RESPONSE_ADAPTER: Final = TypeAdapter(dict[str, object])
_BODY_FRAMING_HEADERS: Final = frozenset({"content-encoding", "content-length"})
_SWARM_DELTA_EVENT: Final = "response.swarm.llm_delta"
class _SwarmChannel(NamedTuple):
"""How one `swarm.data.channel` maps onto the OpenAI event that carries it."""
event_type: str
item_id_prefix: str
index_field: str
# A Deep Research run streams several agents. Only the reporter's `output_text` is
# the answer that lands in the final `response.completed` snapshot; the worker's
# channel-less deltas are an intermediate draft and must not be mistaken for it.
_SWARM_CHANNELS: Final = MappingProxyType(
{
"output_text": _SwarmChannel("response.output_text.delta", "msg", "content_index"),
"reasoning": _SwarmChannel("response.reasoning_summary_text.delta", "rs", "summary_index"),
}
)
class ApodexResponsesConfig(OpenAIResponsesAPIConfig):
@property
def custom_llm_provider(self) -> LlmProviders:
return LlmProviders.APODEX
def validate_environment(
self,
headers: dict, # mutable-ok: matches the base-class signature
model: str,
litellm_params: GenericLiteLLMParams | None,
) -> dict: # mutable-ok: matches the base-class signature
"""Resolve the Apodex key rather than inheriting OpenAI's OPENAI_API_KEY fallback,
which would otherwise forward an unrelated OpenAI key to Apodex."""
resolved_params: Final = litellm_params or GenericLiteLLMParams()
api_key: Final = get_apodex_api_key(resolved_params.api_key)
if api_key is None:
return headers
return { # mutable-ok: matches the base-class signature
**headers,
"Content-Type": "application/json",
"Authorization": f"Bearer {api_key}",
}
def get_complete_url(
self,
api_base: str | None,
litellm_params: dict, # mutable-ok: matches the base-class signature
) -> str:
resolved_base: Final = get_apodex_api_base(api_base).rstrip("/")
return f"{resolved_base}/responses"
def transform_cancel_response_api_response(
self,
raw_response: httpx.Response,
logging_obj: LiteLLMLoggingObj,
) -> ResponsesAPIResponse:
"""Backfill the fields Apodex omits from a cancel payload but ResponsesAPIResponse requires."""
try:
payload: Final = _CANCEL_RESPONSE_ADAPTER.validate_json(raw_response.content)
except ValidationError:
return super().transform_cancel_response_api_response(
raw_response=raw_response,
logging_obj=logging_obj,
)
normalized_response: Final = httpx.Response(
status_code=raw_response.status_code,
# Content-Encoding and Content-Length describe the body being replaced here;
# carrying them over makes httpx try to decompress plain JSON on read.
headers={ # mutable-ok: httpx requires mutable response headers
name: value for name, value in raw_response.headers.items() if name.lower() not in _BODY_FRAMING_HEADERS
},
json={ # mutable-ok: httpx requires a JSON-compatible response dict
**payload,
"created_at": payload.get("created_at", int(time())),
"output": payload.get("output", []), # mutable-ok: Responses payload requires an array default
},
)
return super().transform_cancel_response_api_response(
raw_response=normalized_response,
logging_obj=logging_obj,
)
def transform_streaming_response(
self,
model: str,
parsed_chunk: dict, # mutable-ok: matches the base-class signature
logging_obj: LiteLLMLoggingObj,
) -> ResponsesAPIStreamingResponse:
"""Surface a Deep Research run's text as the OpenAI delta events callers expect.
Observed live, not documented: the stream carries all of its text in
`response.swarm.llm_delta` and never emits `response.output_text.delta` or
any reasoning event, so without this the answer arrives only in the final
`response.completed` snapshot and the reasoning is lost. Everything this
does not recognise, the remaining `response.swarm.*` lifecycle events
included, falls through to the base class as a GenericEvent.
"""
mapped: Final = self._map_swarm_delta(parsed_chunk)
return super().transform_streaming_response(
model=model,
parsed_chunk=parsed_chunk if mapped is None else mapped,
logging_obj=logging_obj,
)
@staticmethod
def _map_swarm_delta(
parsed_chunk: Mapping[str, object],
) -> dict[str, object] | None: # mutable-ok: feeds the base class's `parsed_chunk: dict`
"""The OpenAI event for this swarm delta, or None to pass the chunk through."""
if parsed_chunk.get("type") != _SWARM_DELTA_EVENT:
return None
swarm: Final = parsed_chunk.get("swarm")
data: Final = swarm.get("data") if isinstance(swarm, Mapping) else None
if not isinstance(data, Mapping):
return None
channel_name: Final = data.get("channel")
delta: Final = data.get("delta")
if not isinstance(channel_name, str):
return None
channel: Final = _SWARM_CHANNELS.get(channel_name)
if channel is None or not isinstance(delta, str):
return None
response_id: Final = str(parsed_chunk.get("response_id", ""))
return { # mutable-ok: JSON event payload
"type": channel.event_type,
"item_id": f"{channel.item_id_prefix}_{response_id}",
"output_index": 0,
channel.index_field: 0,
"delta": delta,
"sequence_number": parsed_chunk.get("sequence_number", 0),
}
def get_supported_openai_params(self, model: str) -> list: # mutable-ok: matches the base-class signature
inherited: Final = super().get_supported_openai_params(model)
if is_deep_research_model(model):
return inherited
return [ # mutable-ok: matches the base-class signature
param for param in inherited if param not in _STATEFUL_PARAMS
]
def map_openai_params(
self,
response_api_optional_params: ResponsesAPIOptionalRequestParams,
model: str,
drop_params: bool,
) -> dict: # mutable-ok: matches the base-class signature
mapped: Final = super().map_openai_params(
response_api_optional_params=response_api_optional_params,
model=model,
drop_params=drop_params,
)
stateless: Final = (
mapped
if is_deep_research_model(model)
else self._enforce_stateless(mapped, model=model, drop_params=drop_params)
)
if stateless.get("stream"):
return {**stateless} # mutable-ok: JSON request body
return {**stateless, "stream": False} # mutable-ok: JSON request body
@staticmethod
def _enforce_stateless(params: Mapping[str, object], model: str, drop_params: bool) -> Mapping[str, object]:
"""Core models only: drop what the stateless subset rejects and pin store to false."""
if params.get("store") is True and not (drop_params or litellm.drop_params):
raise litellm.UnsupportedParamsError(
message=(
f"apodex model {model} does not support store=True on /v1/responses: the endpoint is a "
"stateless subset. To drop this, set `litellm.drop_params = True`"
),
status_code=400,
)
kept: Final = { # mutable-ok: JSON request body
key: value for key, value in params.items() if key not in _STATEFUL_PARAMS
}
return {**kept, "store": False} # mutable-ok: JSON request body

View file

@ -50978,6 +50978,128 @@
],
"supports_audio_output": true
},
"apodex/apodex-1.1": {
"max_tokens": 65536,
"max_input_tokens": 262144,
"max_output_tokens": 65536,
"input_cost_per_token": 3e-07,
"cache_read_input_token_cost": 3e-08,
"output_cost_per_token": 3e-06,
"input_cost_per_token_above_200k_tokens": 6e-07,
"cache_read_input_token_cost_above_200k_tokens": 6e-08,
"output_cost_per_token_above_200k_tokens": 6e-06,
"litellm_provider": "apodex",
"mode": "chat",
"source": "https://platform.apodex.ai/docs/models",
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses",
"/v1/messages"
],
"supports_function_calling": true,
"supports_native_streaming": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": false,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": false
},
"apodex/apodex-1.1-mini": {
"max_tokens": 65536,
"max_input_tokens": 262144,
"max_output_tokens": 65536,
"input_cost_per_token": 1e-07,
"cache_read_input_token_cost": 1e-08,
"output_cost_per_token": 1e-06,
"input_cost_per_token_above_200k_tokens": 2e-07,
"cache_read_input_token_cost_above_200k_tokens": 2e-08,
"output_cost_per_token_above_200k_tokens": 2e-06,
"litellm_provider": "apodex",
"mode": "chat",
"source": "https://platform.apodex.ai/docs/models",
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses",
"/v1/messages"
],
"supports_function_calling": true,
"supports_native_streaming": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": false,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": false
},
"apodex/apodex-1-1-deep-research": {
"max_tokens": 65536,
"max_input_tokens": 131072,
"max_output_tokens": 65536,
"input_cost_per_token": 5e-06,
"output_cost_per_token": 2e-05,
"litellm_provider": "apodex",
"mode": "chat",
"source": "https://platform.apodex.ai/docs/pricing",
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
],
"supports_function_calling": false,
"supports_native_streaming": true,
"supports_prompt_caching": false,
"supports_reasoning": true,
"supports_response_schema": false,
"supports_system_messages": true,
"supports_tool_choice": false,
"supports_vision": false,
"supports_web_search": true
},
"apodex/apodex-1-1-deep-solve": {
"max_tokens": 65536,
"max_input_tokens": 131072,
"max_output_tokens": 65536,
"input_cost_per_token": 5e-06,
"output_cost_per_token": 2.5e-05,
"litellm_provider": "apodex",
"mode": "chat",
"source": "https://platform.apodex.ai/docs/pricing",
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
],
"supports_function_calling": false,
"supports_native_streaming": true,
"supports_prompt_caching": false,
"supports_reasoning": true,
"supports_response_schema": false,
"supports_system_messages": true,
"supports_tool_choice": false,
"supports_vision": false,
"supports_web_search": true
},
"apodex/apodex-1-1-deep-discover": {
"max_tokens": 262144,
"max_input_tokens": 131072,
"max_output_tokens": 262144,
"input_cost_per_token": 1e-05,
"output_cost_per_token": 0.0001,
"litellm_provider": "apodex",
"mode": "chat",
"source": "https://platform.apodex.ai/docs/pricing",
"supported_endpoints": [
"/v1/responses"
],
"supports_function_calling": false,
"supports_native_streaming": true,
"supports_prompt_caching": false,
"supports_reasoning": true,
"supports_response_schema": false,
"supports_system_messages": true,
"supports_tool_choice": false,
"supports_vision": false,
"supports_web_search": true
},
"fallback_generalizations": {
"rules": [
{

View file

@ -177,6 +177,23 @@
"interactions": true
}
},
"apodex": {
"display_name": "Apodex (`apodex`)",
"url": "https://docs.litellm.ai/docs/providers/apodex",
"endpoints": {
"chat_completions": true,
"messages": true,
"responses": true,
"embeddings": false,
"image_generations": false,
"audio_transcriptions": false,
"audio_speech": false,
"moderations": false,
"batches": false,
"rerank": false,
"a2a": false
}
},
"apertis": {
"display_name": "Apertis (`apertis`)",
"endpoints": {

View file

@ -3809,6 +3809,7 @@ class LlmProviders(str, Enum):
SCX_AI = "scx-ai"
DARKBLOOM = "darkbloom"
META = "meta"
APODEX = "apodex"
LITELLM_AGENT = "litellm_agent"
CURSOR = "cursor"
BEDROCK_MANTLE = "bedrock_mantle"

View file

@ -8067,6 +8067,7 @@ class ProviderConfigManager:
),
LlmProviders.GRADIENT_AI: (lambda: litellm.GradientAIConfig(), False),
LlmProviders.NSCALE: (lambda: litellm.NscaleConfig(), False),
LlmProviders.APODEX: (lambda: litellm.ApodexChatConfig(), False),
LlmProviders.HEROKU: (lambda: litellm.HerokuChatConfig(), False),
LlmProviders.OCI: (lambda: litellm.OCIChatConfig(), False),
LlmProviders.HYPERBOLIC: (lambda: litellm.HyperbolicChatConfig(), False),
@ -8394,6 +8395,17 @@ class ProviderConfigManager:
)
return DeepSeekAnthropicMessagesConfig()
elif litellm.LlmProviders.APODEX == provider:
from litellm.llms.apodex.common_utils import is_deep_research_model
from litellm.llms.apodex.messages.transformation import (
ApodexAnthropicMessagesConfig,
)
# Apodex only serves the core models on its native /v1/messages path; the
# deep research tiers get no config so they fall back to translation.
if is_deep_research_model(model):
return None
return ApodexAnthropicMessagesConfig()
elif litellm.LlmProviders.TENCENT == provider:
from litellm.llms.tencent.messages.transformation import (
TencentAnthropicMessagesConfig,
@ -8579,6 +8591,8 @@ class ProviderConfigManager:
return litellm.ManusResponsesAPIConfig()
elif litellm.LlmProviders.PERPLEXITY == provider:
return litellm.PerplexityResponsesConfig()
elif litellm.LlmProviders.APODEX == provider:
return litellm.ApodexResponsesConfig()
elif litellm.LlmProviders.DATABRICKS == provider:
# Databricks Responses API is only compatible with OpenAI GPT models
if model and "gpt" in model.lower():

View file

@ -50978,6 +50978,128 @@
],
"supports_audio_output": true
},
"apodex/apodex-1.1": {
"max_tokens": 65536,
"max_input_tokens": 262144,
"max_output_tokens": 65536,
"input_cost_per_token": 3e-07,
"cache_read_input_token_cost": 3e-08,
"output_cost_per_token": 3e-06,
"input_cost_per_token_above_200k_tokens": 6e-07,
"cache_read_input_token_cost_above_200k_tokens": 6e-08,
"output_cost_per_token_above_200k_tokens": 6e-06,
"litellm_provider": "apodex",
"mode": "chat",
"source": "https://platform.apodex.ai/docs/models",
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses",
"/v1/messages"
],
"supports_function_calling": true,
"supports_native_streaming": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": false,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": false
},
"apodex/apodex-1.1-mini": {
"max_tokens": 65536,
"max_input_tokens": 262144,
"max_output_tokens": 65536,
"input_cost_per_token": 1e-07,
"cache_read_input_token_cost": 1e-08,
"output_cost_per_token": 1e-06,
"input_cost_per_token_above_200k_tokens": 2e-07,
"cache_read_input_token_cost_above_200k_tokens": 2e-08,
"output_cost_per_token_above_200k_tokens": 2e-06,
"litellm_provider": "apodex",
"mode": "chat",
"source": "https://platform.apodex.ai/docs/models",
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses",
"/v1/messages"
],
"supports_function_calling": true,
"supports_native_streaming": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": false,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": false
},
"apodex/apodex-1-1-deep-research": {
"max_tokens": 65536,
"max_input_tokens": 131072,
"max_output_tokens": 65536,
"input_cost_per_token": 5e-06,
"output_cost_per_token": 2e-05,
"litellm_provider": "apodex",
"mode": "chat",
"source": "https://platform.apodex.ai/docs/pricing",
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
],
"supports_function_calling": false,
"supports_native_streaming": true,
"supports_prompt_caching": false,
"supports_reasoning": true,
"supports_response_schema": false,
"supports_system_messages": true,
"supports_tool_choice": false,
"supports_vision": false,
"supports_web_search": true
},
"apodex/apodex-1-1-deep-solve": {
"max_tokens": 65536,
"max_input_tokens": 131072,
"max_output_tokens": 65536,
"input_cost_per_token": 5e-06,
"output_cost_per_token": 2.5e-05,
"litellm_provider": "apodex",
"mode": "chat",
"source": "https://platform.apodex.ai/docs/pricing",
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
],
"supports_function_calling": false,
"supports_native_streaming": true,
"supports_prompt_caching": false,
"supports_reasoning": true,
"supports_response_schema": false,
"supports_system_messages": true,
"supports_tool_choice": false,
"supports_vision": false,
"supports_web_search": true
},
"apodex/apodex-1-1-deep-discover": {
"max_tokens": 262144,
"max_input_tokens": 131072,
"max_output_tokens": 262144,
"input_cost_per_token": 1e-05,
"output_cost_per_token": 0.0001,
"litellm_provider": "apodex",
"mode": "chat",
"source": "https://platform.apodex.ai/docs/pricing",
"supported_endpoints": [
"/v1/responses"
],
"supports_function_calling": false,
"supports_native_streaming": true,
"supports_prompt_caching": false,
"supports_reasoning": true,
"supports_response_schema": false,
"supports_system_messages": true,
"supports_tool_choice": false,
"supports_vision": false,
"supports_web_search": true
},
"fallback_generalizations": {
"rules": [
{

View file

@ -177,6 +177,23 @@
"interactions": true
}
},
"apodex": {
"display_name": "Apodex (`apodex`)",
"url": "https://docs.litellm.ai/docs/providers/apodex",
"endpoints": {
"chat_completions": true,
"messages": true,
"responses": true,
"embeddings": false,
"image_generations": false,
"audio_transcriptions": false,
"audio_speech": false,
"moderations": false,
"batches": false,
"rerank": false,
"a2a": false
}
},
"apertis": {
"display_name": "Apertis (`apertis`)",
"endpoints": {

View file

@ -245,10 +245,16 @@ class TestProcessEventTextDeltaWithoutOutputItemAdded:
assert chunks[1]["delta"] == {"type": "text_delta", "text": "Hel"}
def test_process_event_delta_without_item_id_never_yields_negative_index(self):
chunks = _process_all([{"type": "response.output_text.delta", "delta": "Hi"}])
chunks = _process_all(
[
{"type": "response.output_text.delta", "delta": "Hi"},
{"type": "response.output_text.delta", "delta": " again"},
]
)
assert [(c["type"], c["index"]) for c in chunks] == [
("content_block_start", 0),
("content_block_delta", 0),
("content_block_delta", 0),
]
def test_process_event_unregistered_item_id_opens_new_text_block(self):
@ -262,9 +268,10 @@ class TestProcessEventTextDeltaWithoutOutputItemAdded:
{"type": "response.output_text.delta", "item_id": "m1", "delta": "Hi"},
]
)
assert chunks[2]["type"] == "content_block_start"
assert chunks[2]["content_block"] == {"type": "text", "text": ""}
assert [c["index"] for c in chunks[2:]] == [1, 1]
assert chunks[2] == {"type": "content_block_stop", "index": 0}
assert chunks[3]["type"] == "content_block_start"
assert chunks[3]["content_block"] == {"type": "text", "text": ""}
assert [c["index"] for c in chunks[2:]] == [0, 1, 1]
def test_process_event_registered_item_id_does_not_synthesize_start(self):
chunks = _process_all(
@ -282,6 +289,91 @@ class TestProcessEventTextDeltaWithoutOutputItemAdded:
]
class TestProcessEventReasoningDeltaWithoutOutputItemAdded:
def test_reasoning_opens_thinking_block_before_delta(self):
chunks = _process_all(
[
{"type": "response.reasoning_summary_text.delta", "item_id": "rs_1", "delta": "Think "},
{"type": "response.reasoning_summary_text.delta", "item_id": "rs_1", "delta": "carefully"},
]
)
assert [chunk["type"] for chunk in chunks] == [
"content_block_start",
"content_block_delta",
"content_block_delta",
]
assert chunks[0] == {
"type": "content_block_start",
"index": 0,
"content_block": {"type": "thinking", "thinking": "", "signature": ""},
}
assert chunks[1] == {
"type": "content_block_delta",
"index": 0,
"delta": {"type": "thinking_delta", "thinking": "Think "},
}
def test_reasoning_and_text_get_separate_closed_blocks(self):
response = SimpleNamespace(status="completed", output=[], usage=None)
chunks = _process_all(
[
{"type": "response.reasoning_summary_text.delta", "item_id": "rs_1", "delta": "Think"},
{"type": "response.output_text.delta", "item_id": "msg_1", "delta": "Answer"},
{"type": "response.completed", "response": response},
]
)
assert [(chunk["type"], chunk.get("index")) for chunk in chunks] == [
("content_block_start", 0),
("content_block_delta", 0),
("content_block_stop", 0),
("content_block_start", 1),
("content_block_delta", 1),
("content_block_stop", 1),
("message_delta", None),
("message_stop", None),
]
assert chunks[3]["content_block"] == {"type": "text", "text": ""}
def test_resumed_item_id_opens_a_fresh_block(self):
"""A provider that reuses one item id per run, then interleaves channels, would
otherwise address a delta to a block that has already stopped."""
response = SimpleNamespace(status="completed", output=[], usage=None)
chunks = _process_all(
[
{"type": "response.reasoning_summary_text.delta", "item_id": "rs_1", "delta": "Think A"},
{"type": "response.output_text.delta", "item_id": "msg_1", "delta": "Answer A"},
{"type": "response.reasoning_summary_text.delta", "item_id": "rs_1", "delta": "Think B"},
{"type": "response.output_text.delta", "item_id": "msg_1", "delta": "Answer B"},
{"type": "response.completed", "response": response},
]
)
assert [(chunk["type"], chunk.get("index")) for chunk in chunks] == [
("content_block_start", 0),
("content_block_delta", 0),
("content_block_stop", 0),
("content_block_start", 1),
("content_block_delta", 1),
("content_block_stop", 1),
("content_block_start", 2),
("content_block_delta", 2),
("content_block_stop", 2),
("content_block_start", 3),
("content_block_delta", 3),
("content_block_stop", 3),
("message_delta", None),
("message_stop", None),
]
assert [chunk["content_block"]["type"] for chunk in chunks if chunk["type"] == "content_block_start"] == [
"thinking",
"text",
"thinking",
"text",
]
class TestResponseCompletedUsage:
"""The Anthropic ``message_delta`` usage must report cache reads/writes and
exclude them from ``input_tokens``, so spend is not billed at the uncached

View file

@ -0,0 +1,253 @@
"""
Apodex chat completions transformation.
"""
import json
import httpx
import openai
import pytest
import litellm
from litellm.types.utils import LlmProviders
from litellm.utils import ProviderConfigManager
CORE_MODEL = "apodex/apodex-1.1"
DEEP_RESEARCH_MODEL = "apodex/apodex-1-1-deep-research"
CHAT_RESPONSE = {
"id": "chatcmpl-abc123",
"object": "chat.completion",
"created": 1712345678,
"model": "apodex-1.1",
"choices": [
{
"index": 0,
"message": {"role": "assistant", "content": "ok", "reasoning_content": "let me think"},
"finish_reason": "stop",
}
],
"usage": {
"prompt_tokens": 1000,
"completion_tokens": 100,
"total_tokens": 1100,
"prompt_tokens_details": {"cached_tokens": 500},
},
}
STREAM_BODY = (
b'data: {"id":"chatcmpl-abc123","object":"chat.completion.chunk","created":1,"model":"apodex-1.1",'
b'"choices":[{"index":0,"delta":{"content":"ok"},"finish_reason":null}]}\n\n'
b"data: [DONE]\n\n"
)
@pytest.fixture(autouse=True)
def _apodex_env(monkeypatch: pytest.MonkeyPatch):
"""Resolve models against the in-repo cost map, not the published one."""
monkeypatch.setenv("APODEX_API_KEY", "sk-apodex-test")
monkeypatch.delenv("APODEX_API_BASE", raising=False)
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
yield
def _client(captured: dict, *, stream: bool = False) -> openai.OpenAI:
def handler(request: httpx.Request) -> httpx.Response:
captured["url"] = str(request.url)
captured["body"] = json.loads(request.content)
if stream:
return httpx.Response(200, headers={"content-type": "text/event-stream"}, content=STREAM_BODY)
return httpx.Response(200, json=CHAT_RESPONSE)
return openai.OpenAI(
api_key="sk-apodex-test",
base_url="https://api.apodex.ai/v1",
http_client=httpx.Client(transport=httpx.MockTransport(handler)),
)
def _chat_config(model: str):
return ProviderConfigManager.get_provider_chat_config(model=model, provider=LlmProviders.APODEX)
class TestProviderResolution:
def test_openai_compatible_provider_info_uses_apodex_credentials(self):
config = _chat_config("apodex-1.1")
assert config._get_openai_compatible_provider_info("https://override.test/v1", "sk-override") == (
"https://override.test/v1",
"sk-override",
)
def test_prefixed_model_resolves_to_the_default_base(self):
model, provider, api_key, api_base = litellm.get_llm_provider(model=CORE_MODEL)
assert (model, provider, api_key, api_base) == (
"apodex-1.1",
"apodex",
"sk-apodex-test",
"https://api.apodex.ai/v1",
)
def test_api_base_autodetection(self):
_, provider, api_key, _ = litellm.get_llm_provider(model="apodex-1.1", api_base="https://api.apodex.ai/v1")
assert provider == "apodex"
assert api_key == "sk-apodex-test"
def test_explicit_api_base_and_key_win(self):
_, provider, api_key, api_base = litellm.get_llm_provider(
model=CORE_MODEL, api_base="https://gateway.internal/v1", api_key="sk-override"
)
assert (provider, api_key, api_base) == ("apodex", "sk-override", "https://gateway.internal/v1")
def test_api_base_env_override(self, monkeypatch: pytest.MonkeyPatch):
monkeypatch.setenv("APODEX_API_BASE", "https://env.apodex.test/v1")
_, _, _, api_base = litellm.get_llm_provider(model=CORE_MODEL)
assert api_base == "https://env.apodex.test/v1"
class TestStreamDefault:
"""The Deep Research tiers default `stream` to true, so a non-streaming call must pin it false.
Regression guard: the OpenAI chat handler pops `stream` out of the params it
forwards, which would leave those tiers streaming SSE at a call that cannot
parse it. The core models default to false but are pinned the same way.
"""
def test_non_streaming_call_pins_stream_false(self):
captured: dict = {}
response = litellm.completion(
model=CORE_MODEL,
messages=[{"role": "user", "content": "hi"}],
client=_client(captured),
)
assert captured["url"] == "https://api.apodex.ai/v1/chat/completions"
assert captured["body"]["stream"] is False
assert captured["body"]["model"] == "apodex-1.1"
assert response.choices[0].message.reasoning_content == "let me think"
def test_streaming_call_sends_stream_true(self):
captured: dict = {}
chunks = list(
litellm.completion(
model=CORE_MODEL,
messages=[{"role": "user", "content": "hi"}],
stream=True,
client=_client(captured, stream=True),
)
)
assert captured["body"]["stream"] is True
assert chunks
def test_deep_research_models_pin_stream_too(self):
captured: dict = {}
litellm.completion(
model=DEEP_RESEARCH_MODEL,
messages=[{"role": "user", "content": "hi"}],
client=_client(captured),
)
assert captured["body"]["stream"] is False
def test_user_supplied_extra_body_is_preserved(self):
"""Deep research tiers reach external tools through `mcp_servers` in extra_body."""
captured: dict = {}
mcp_servers = [{"name": "docs", "url": "https://example.com/mcp"}]
litellm.completion(
model=DEEP_RESEARCH_MODEL,
messages=[{"role": "user", "content": "hi"}],
extra_body={"mcp_servers": mcp_servers},
client=_client(captured),
)
assert captured["body"]["stream"] is False
assert captured["body"]["mcp_servers"] == mcp_servers
def test_extra_body_cannot_override_non_streaming_pin(self):
captured: dict = {}
litellm.completion(
model=DEEP_RESEARCH_MODEL,
messages=[{"role": "user", "content": "hi"}],
extra_body={"stream": True},
client=_client(captured),
)
assert captured["body"]["stream"] is False
def test_transform_does_not_mutate_optional_params(self):
config = _chat_config("apodex-1.1")
optional_params = config.map_openai_params(
non_default_params={}, optional_params={}, model="apodex-1.1", drop_params=False
)
original = optional_params.copy()
config.transform_request(
model="apodex-1.1",
messages=[{"role": "user", "content": "hi"}],
optional_params=optional_params,
litellm_params={"custom_llm_provider": "apodex"},
headers={},
)
assert optional_params == original
class TestSupportedParams:
def test_responses_only_model_rejects_chat_completions(self):
with pytest.raises(litellm.BadRequestError, match="only available through /v1/responses"):
litellm.completion(
model="apodex/apodex-1-1-deep-discover",
messages=[{"role": "user", "content": "hi"}],
client=_client({}),
)
def test_core_models_support_tools(self):
supported = _chat_config("apodex-1.1").get_supported_openai_params("apodex-1.1")
assert "tools" in supported
assert "tool_choice" in supported
assert "temperature" in supported
assert "top_p" in supported
def test_deep_research_rejects_tools_and_sampling_params(self):
"""The tiers document tools as unsupported and sampling params as ignored."""
supported = _chat_config(DEEP_RESEARCH_MODEL).get_supported_openai_params("apodex-1-1-deep-research")
for param in ("tools", "tool_choice", "function_call", "functions", "parallel_tool_calls"):
assert param not in supported
assert "temperature" not in supported
assert "top_p" not in supported
assert "max_tokens" in supported
def test_tools_on_a_deep_research_model_raise(self):
with pytest.raises(litellm.UnsupportedParamsError, match="tools"):
litellm.completion(
model=DEEP_RESEARCH_MODEL,
messages=[{"role": "user", "content": "hi"}],
tools=[{"type": "function", "function": {"name": "f", "parameters": {}}}],
client=_client({}),
)
def test_max_completion_tokens_is_renamed_to_max_tokens(self):
captured: dict = {}
litellm.completion(
model=CORE_MODEL,
messages=[{"role": "user", "content": "hi"}],
max_completion_tokens=512,
client=_client(captured),
)
assert captured["body"]["max_tokens"] == 512
assert "max_completion_tokens" not in captured["body"]
class TestCostTracking:
def test_cached_input_is_billed_at_the_lower_rate(self):
captured: dict = {}
response = litellm.completion(
model=CORE_MODEL,
messages=[{"role": "user", "content": "hi"}],
client=_client(captured),
)
# 500 fresh input + 500 cached input + 100 output
expected = 500 * 3e-07 + 500 * 3e-08 + 100 * 3e-06
assert litellm.completion_cost(response, model=CORE_MODEL) == pytest.approx(expected)

View file

@ -0,0 +1,153 @@
"""
Apodex provider registration and model-family classification.
"""
import json
from pathlib import Path
import pytest
import litellm
from litellm.llms.apodex.common_utils import (
APODEX_API_BASE_URL,
get_apodex_api_base,
get_apodex_api_key,
is_deep_research_model,
)
from litellm.types.utils import LlmProviders
REPO_ROOT = Path(__file__).parents[4]
CORE_MODELS = ("apodex-1.1", "apodex-1.1-mini")
DEEP_RESEARCH_MODELS = (
"apodex-1-1-deep-research",
"apodex-1-1-deep-solve",
"apodex-1-1-deep-discover",
)
class TestModelFamily:
"""The model id, not the provider, selects which Apodex contract applies."""
@pytest.mark.parametrize("model", CORE_MODELS)
def test_core_models_are_not_deep_research(self, model: str):
assert is_deep_research_model(model) is False
assert is_deep_research_model(f"apodex/{model}") is False
@pytest.mark.parametrize("model", DEEP_RESEARCH_MODELS)
def test_deep_research_models_are_detected(self, model: str):
assert is_deep_research_model(model) is True
assert is_deep_research_model(f"apodex/{model}") is True
def test_prefix_does_not_leak_into_classification(self):
"""A provider prefix containing the marker must not flip a core model."""
assert is_deep_research_model("some-deep-gateway/apodex-1.1") is False
class TestCredentialResolution:
def test_defaults(self, monkeypatch: pytest.MonkeyPatch):
monkeypatch.delenv("APODEX_API_BASE", raising=False)
monkeypatch.setenv("APODEX_API_KEY", "sk-env")
assert get_apodex_api_base(None) == APODEX_API_BASE_URL
assert get_apodex_api_key(None) == "sk-env"
def test_explicit_values_win_over_env(self, monkeypatch: pytest.MonkeyPatch):
monkeypatch.setenv("APODEX_API_BASE", "https://env.apodex.test/v1")
monkeypatch.setenv("APODEX_API_KEY", "sk-env")
assert get_apodex_api_base("https://explicit.apodex.test/v1") == "https://explicit.apodex.test/v1"
assert get_apodex_api_key("sk-explicit") == "sk-explicit"
def test_env_base_overrides_default(self, monkeypatch: pytest.MonkeyPatch):
monkeypatch.setenv("APODEX_API_BASE", "https://env.apodex.test/v1")
assert get_apodex_api_base(None) == "https://env.apodex.test/v1"
class TestRegistration:
def test_provider_enum_and_lists(self):
assert LlmProviders.APODEX.value == "apodex"
assert "apodex" in litellm.provider_list
assert "apodex" in litellm.constants.openai_compatible_providers
assert APODEX_API_BASE_URL in litellm.constants.openai_compatible_endpoints
def test_not_registered_as_a_json_provider(self):
"""Apodex needs model-aware transformations, so it must not fall into the
generic JSON path, which would shadow the Python configs in provider resolution."""
from litellm.llms.openai_like.json_loader import JSONProviderRegistry
assert JSONProviderRegistry.exists("apodex") is False
def test_config_classes_resolve_from_the_lazy_registry(self):
assert litellm.ApodexChatConfig().custom_llm_provider == "apodex"
assert litellm.ApodexResponsesConfig().custom_llm_provider == LlmProviders.APODEX
def test_packaged_endpoint_matrix_matches_the_source(self):
source = json.loads((REPO_ROOT / "provider_endpoints_support.json").read_text())
backup = json.loads((REPO_ROOT / "litellm" / "provider_endpoints_support_backup.json").read_text())
assert backup["providers"]["apodex"] == source["providers"]["apodex"]
class TestModelMetadata:
@pytest.fixture(scope="class")
def model_cost(self) -> dict:
with open(REPO_ROOT / "model_prices_and_context_window.json") as f:
return json.load(f)
def test_every_apodex_model_is_registered(self, model_cost: dict):
assert {key for key in model_cost if key.startswith("apodex/")} == {
f"apodex/{model}" for model in (*CORE_MODELS, *DEEP_RESEARCH_MODELS)
}
def test_core_model_pricing(self, model_cost: dict):
info = model_cost["apodex/apodex-1.1"]
assert info["litellm_provider"] == "apodex"
assert info["mode"] == "chat"
assert info["max_input_tokens"] == 262144
# GET /v1/models reports max_completion_tokens 65536, well under the context window
assert info["max_output_tokens"] == 65536
assert info["max_tokens"] == info["max_output_tokens"]
assert info["input_cost_per_token"] == 3e-07
assert info["cache_read_input_token_cost"] == 3e-08
assert info["output_cost_per_token"] == 3e-06
# Requests over 200K input tokens are billed at 2x across every tier
assert info["input_cost_per_token_above_200k_tokens"] == 6e-07
assert info["cache_read_input_token_cost_above_200k_tokens"] == 6e-08
assert info["output_cost_per_token_above_200k_tokens"] == 6e-06
assert info["supports_prompt_caching"] is True
assert info["supports_function_calling"] is True
assert info["supported_endpoints"] == ["/v1/chat/completions", "/v1/responses", "/v1/messages"]
def test_deep_research_model_pricing(self, model_cost: dict):
info = model_cost["apodex/apodex-1-1-deep-research"]
assert info["max_input_tokens"] == 131072
assert info["max_output_tokens"] == 65536
assert info["input_cost_per_token"] == 5e-06
assert info["output_cost_per_token"] == 2e-05
assert info["supports_function_calling"] is False
assert info["supports_response_schema"] is False
assert info["supports_prompt_caching"] is False
assert info["supports_web_search"] is True
@pytest.mark.parametrize("model", DEEP_RESEARCH_MODELS)
def test_deep_research_models_are_not_on_the_native_messages_path(self, model_cost: dict, model: str):
"""Apodex serves /v1/messages for the core models only."""
assert "/v1/messages" not in model_cost[f"apodex/{model}"]["supported_endpoints"]
def test_discover_is_responses_only(self, model_cost: dict):
"""The Discover tiers answer 400 unsupported_api on /v1/chat/completions."""
assert model_cost["apodex/apodex-1-1-deep-discover"]["supported_endpoints"] == ["/v1/responses"]
@pytest.mark.parametrize("model", ("apodex-1-1-deep-research", "apodex-1-1-deep-solve"))
def test_the_other_deep_tiers_keep_chat_completions(self, model_cost: dict, model: str):
assert model_cost[f"apodex/{model}"]["supported_endpoints"] == [
"/v1/chat/completions",
"/v1/responses",
]
def test_backup_cost_map_in_sync(self, model_cost: dict):
with open(REPO_ROOT / "litellm" / "model_prices_and_context_window_backup.json") as f:
backup = json.load(f)
for key in (key for key in model_cost if key.startswith("apodex/")):
assert backup[key] == model_cost[key], f"{key} differs between main and backup cost maps"

View file

@ -0,0 +1,128 @@
"""
Apodex Anthropic Messages transformation.
Apodex implements the Anthropic protocol natively at POST /v1/messages, but only
serves the core models there. The Deep Research tiers must keep working on the
same route through LiteLLM's translation instead of being handed to a path that
would reject them.
"""
import pytest
import litellm
from litellm.types.utils import LlmProviders
from litellm.utils import ProviderConfigManager
CORE_MODELS = ("apodex-1.1", "apodex-1.1-mini")
DEEP_RESEARCH_MODELS = (
"apodex-1-1-deep-research",
"apodex-1-1-deep-solve",
"apodex-1-1-deep-discover",
)
@pytest.fixture(autouse=True)
def _apodex_env(monkeypatch: pytest.MonkeyPatch):
monkeypatch.setenv("APODEX_API_KEY", "sk-apodex-test")
monkeypatch.delenv("APODEX_API_BASE", raising=False)
yield
def _messages_config(model: str):
return ProviderConfigManager.get_provider_anthropic_messages_config(model=model, provider=LlmProviders.APODEX)
def _complete_url() -> str:
"""Resolve the endpoint the way the handler does: validate first, then build the URL.
validate_anthropic_messages_environment returns the api_base the handler feeds
into get_complete_url, so the two steps have to run in that order.
"""
config = _messages_config("apodex-1.1")
assert config is not None
_, api_base = config.validate_anthropic_messages_environment(
headers={}, model="apodex-1.1", messages=[], optional_params={}, litellm_params={}
)
return config.get_complete_url(
api_base=api_base, api_key=None, model="apodex-1.1", optional_params={}, litellm_params={}
)
class TestNativePassthroughRouting:
@pytest.mark.parametrize("model", CORE_MODELS)
def test_core_models_get_the_native_config(self, model: str):
config = _messages_config(model)
assert config is not None
assert type(config).__name__ == "ApodexAnthropicMessagesConfig"
assert config.custom_llm_provider == "apodex"
assert config.should_strip_billing_metadata() is True
@pytest.mark.parametrize("model", DEEP_RESEARCH_MODELS)
def test_deep_research_models_fall_back_to_translation(self, model: str):
"""No native config means LiteLLM uses a protocol translation instead of
forwarding to a path Apodex does not serve for these tiers."""
assert _messages_config(model) is None
@pytest.mark.parametrize("stream", (False, True), ids=("non-streaming", "streaming"))
def test_deep_research_translation_uses_responses_api(self, monkeypatch: pytest.MonkeyPatch, stream: bool):
from litellm.llms.anthropic.experimental_pass_through.messages import handler
captured: dict = {}
class ResponsesRouteSelected(Exception):
pass
def capture_responses_translation(**kwargs):
captured.update(kwargs)
raise ResponsesRouteSelected
def reject_chat_translation(**kwargs):
pytest.fail("Apodex Deep Research messages must not route through chat completions")
monkeypatch.setattr(litellm, "responses", capture_responses_translation)
monkeypatch.setattr(litellm, "completion", reject_chat_translation)
with pytest.raises(ResponsesRouteSelected):
handler.anthropic_messages_handler(
max_tokens=256,
messages=[{"role": "user", "content": "hi"}],
model="apodex/apodex-1-1-deep-research",
custom_llm_provider="apodex",
stream=stream,
)
assert captured["model"] == "apodex-1-1-deep-research"
assert captured["custom_llm_provider"] == "apodex"
assert captured.get("stream", False) is stream
class TestNativePassthroughRequest:
def test_url_targets_the_native_messages_path(self):
assert _complete_url() == "https://api.apodex.ai/v1/messages"
def test_url_honours_an_api_base_override(self, monkeypatch: pytest.MonkeyPatch):
monkeypatch.setenv("APODEX_API_BASE", "https://env.apodex.test/v1")
assert _complete_url() == "https://env.apodex.test/v1/messages"
def test_headers_use_the_provider_api_key(self):
config = _messages_config("apodex-1.1")
assert config is not None
headers, _ = config.validate_anthropic_messages_environment(
headers={}, model="apodex-1.1", messages=[], optional_params={}, litellm_params={}
)
assert headers["authorization"] == "Bearer sk-apodex-test"
assert headers["anthropic-version"] == "2023-06-01"
assert headers["content-type"] == "application/json"
def test_caller_supplied_auth_header_is_not_overwritten(self):
config = _messages_config("apodex-1.1")
assert config is not None
headers, _ = config.validate_anthropic_messages_environment(
headers={"x-api-key": "sk-caller"},
model="apodex-1.1",
messages=[],
optional_params={},
litellm_params={},
)
assert headers["x-api-key"] == "sk-caller"
assert "authorization" not in headers

View file

@ -0,0 +1,426 @@
"""
Apodex Responses API transformation.
The core models expose a stateless subset of /v1/responses while the Deep
Research tiers keep server-side state, so the parameter contract is keyed off
the model rather than applied provider-wide.
"""
import gzip
from types import SimpleNamespace
import httpx
import pytest
import litellm
from litellm.llms.anthropic.experimental_pass_through.responses_adapters.streaming_iterator import (
AnthropicResponsesStreamWrapper,
)
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler
from litellm.types.utils import LlmProviders
from litellm.utils import ProviderConfigManager
CORE_MODEL = "apodex/apodex-1.1"
CORE_MINI_MODEL = "apodex/apodex-1.1-mini"
DEEP_RESEARCH_MODEL = "apodex/apodex-1-1-deep-research"
@pytest.fixture(autouse=True)
def _apodex_env(monkeypatch: pytest.MonkeyPatch):
monkeypatch.setenv("APODEX_API_KEY", "sk-apodex-test")
monkeypatch.delenv("APODEX_API_BASE", raising=False)
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
monkeypatch.setattr(litellm, "drop_params", False)
yield
_SENTINEL = "apodex-request-captured"
def _capture(**kwargs) -> dict:
"""Run litellm.responses() and return the request it would have sent.
Validation errors raised before the request is built propagate to the caller.
"""
captured: dict = {}
class CapturingHandler(HTTPHandler):
def post(self, *args, **post_kwargs):
captured.update(url=post_kwargs.get("url"), body=post_kwargs.get("json"))
raise RuntimeError(_SENTINEL)
try:
litellm.responses(client=CapturingHandler(), **kwargs)
except Exception as exc:
if _SENTINEL not in str(exc):
raise
assert captured, "no request was sent"
return captured
def _responses_config(model: str):
return ProviderConfigManager.get_provider_responses_api_config(model=model, provider=LlmProviders.APODEX)
class TestConfigSelection:
def test_python_config_is_used_for_every_apodex_model(self):
for model in ("apodex-1.1", "apodex-1.1-mini", "apodex-1-1-deep-research"):
config = _responses_config(model)
assert type(config).__name__ == "ApodexResponsesConfig"
def test_auth_uses_the_apodex_key(self):
config = _responses_config("apodex-1.1")
assert config.validate_environment(headers={}, model="apodex-1.1", litellm_params=None) == {
"Content-Type": "application/json",
"Authorization": "Bearer sk-apodex-test",
}
def test_auth_does_not_fall_back_to_an_openai_key(self, monkeypatch: pytest.MonkeyPatch):
"""The inherited OpenAI config would forward OPENAI_API_KEY to Apodex."""
monkeypatch.delenv("APODEX_API_KEY", raising=False)
monkeypatch.setenv("OPENAI_API_KEY", "sk-openai-must-not-leak")
config = _responses_config("apodex-1.1")
assert config.validate_environment(headers={}, model="apodex-1.1", litellm_params=None) == {}
def test_request_targets_the_apodex_responses_url(self):
assert _capture(model=CORE_MODEL, input="hi")["url"] == "https://api.apodex.ai/v1/responses"
def test_polling_without_model_resolution_targets_apodex(self):
config = _responses_config("apodex-1-1-deep-research")
assert config.get_complete_url(api_base=None, litellm_params={}) == "https://api.apodex.ai/v1/responses"
def test_polling_honours_an_explicit_api_base(self):
config = _responses_config("apodex-1-1-deep-research")
assert (
config.get_complete_url(api_base="https://gateway.apodex.test/v1/", litellm_params={})
== "https://gateway.apodex.test/v1/responses"
)
def test_request_honours_an_api_base_override(self, monkeypatch: pytest.MonkeyPatch):
monkeypatch.setenv("APODEX_API_BASE", "https://env.apodex.test/v1")
assert _capture(model=CORE_MODEL, input="hi")["url"] == "https://env.apodex.test/v1/responses"
class TestStreamDefault:
def test_non_streaming_pins_stream_false(self):
captured = _capture(model=DEEP_RESEARCH_MODEL, input="hi")
assert captured["url"] == "https://api.apodex.ai/v1/responses"
assert captured["body"]["stream"] is False
@pytest.mark.asyncio
async def test_streaming_sends_stream_true(self):
captured: dict = {}
class CapturingHandler(AsyncHTTPHandler):
async def post(self, *args, **kwargs):
captured.update(body=kwargs.get("json"))
raise RuntimeError("captured")
with pytest.raises(Exception, match="captured"):
await litellm.aresponses(model=DEEP_RESEARCH_MODEL, input="hi", stream=True, client=CapturingHandler())
assert captured["body"]["stream"] is True
class TestCoreModelStatelessSubset:
"""Apodex core models reject anything that would persist state on their side."""
@pytest.mark.parametrize("model", [CORE_MODEL, CORE_MINI_MODEL])
def test_store_is_pinned_false(self, model: str):
captured = _capture(model=model, input="hi")
assert captured["body"]["store"] is False
def test_store_true_raises(self):
with pytest.raises(litellm.UnsupportedParamsError, match="store=True"):
_capture(model=CORE_MODEL, input="hi", store=True)
def test_background_raises(self):
with pytest.raises(litellm.UnsupportedParamsError, match="background"):
_capture(model=CORE_MODEL, input="hi", background=True)
def test_previous_response_id_raises(self):
with pytest.raises(litellm.UnsupportedParamsError, match="previous_response_id"):
_capture(model=CORE_MODEL, input="hi", previous_response_id="resp_1")
def test_drop_params_strips_all_three(self, monkeypatch: pytest.MonkeyPatch):
monkeypatch.setattr(litellm, "drop_params", True)
captured = _capture(
model=CORE_MODEL,
input="hi",
store=True,
background=True,
previous_response_id="resp_1",
)
assert captured["body"]["store"] is False
assert "background" not in captured["body"]
assert "previous_response_id" not in captured["body"]
def test_stateful_params_are_not_advertised(self):
supported = _responses_config("apodex-1.1").get_supported_openai_params("apodex-1.1")
assert "background" not in supported
assert "previous_response_id" not in supported
assert "max_output_tokens" in supported
def test_max_output_tokens_still_passes_through(self):
captured = _capture(model=CORE_MODEL, input="hi", max_output_tokens=512)
assert captured["body"]["max_output_tokens"] == 512
class TestDeepResearchKeepsState:
"""The agent tiers survive client disconnects, so none of this may be stripped."""
def test_background_passes_through(self):
captured = _capture(model=DEEP_RESEARCH_MODEL, input="hi", background=True)
assert captured["body"]["background"] is True
def test_store_and_previous_response_id_pass_through(self):
captured = _capture(model=DEEP_RESEARCH_MODEL, input="hi", store=True, previous_response_id="resp_1")
assert captured["body"]["store"] is True
assert captured["body"]["previous_response_id"] == "resp_1"
def test_store_is_not_pinned_when_unset(self):
captured = _capture(model=DEEP_RESEARCH_MODEL, input="hi")
assert "store" not in captured["body"]
def test_stateful_params_are_advertised(self):
supported = _responses_config("apodex-1-1-deep-research").get_supported_openai_params(
"apodex-1-1-deep-research"
)
assert "background" in supported
assert "previous_response_id" in supported
def test_minimal_cancel_response_is_normalized(self):
config = _responses_config("apodex-1-1-deep-research")
response = config.transform_cancel_response_api_response(
raw_response=httpx.Response(
200,
json={"id": "resp_1", "object": "response", "status": "cancelled"},
),
logging_obj=None,
)
assert response.id == "resp_1"
assert response.status == "cancelled"
assert response.output == []
assert response.created_at > 0
def test_cancel_response_survives_a_compressed_upstream_response(self):
"""The body is rebuilt, so the original framing headers must not follow it.
httpx decodes on read, so carrying Content-Encoding over from the compressed
upstream response makes it try to gunzip the plain JSON replacement.
"""
body = b'{"id": "resp_1", "object": "response", "status": "cancelled"}'
# As it arrives off the wire: httpx decodes the body but leaves the header in place
upstream = httpx.Response(
200,
headers={
"content-encoding": "gzip",
"x-ratelimit-remaining-requests": "42",
},
content=gzip.compress(body),
)
assert upstream.content == body
config = _responses_config("apodex-1-1-deep-research")
response = config.transform_cancel_response_api_response(
raw_response=upstream,
logging_obj=None,
)
assert response.id == "resp_1"
assert response.status == "cancelled"
assert response._hidden_params["headers"]["x-ratelimit-remaining-requests"] == "42"
def test_output_text_delta_is_normalized(self):
"""A Deep Research stream never emits response.output_text.delta of its own.
Payload shape captured from a live stream; the extra swarm.data keys ride
along untouched and must not affect the mapping.
"""
config = _responses_config("apodex-1-1-deep-research")
event = config.transform_streaming_response(
model="apodex-1-1-deep-research",
parsed_chunk={
"type": "response.swarm.llm_delta",
"created_at": 1786873219.3543231,
"response_id": "w_c4b77c96",
"sequence_number": 12,
"swarm": {
"agent_id": "reporter",
"data": {
"channel": "output_text",
"delta": "Hello there, friend!",
"delta_index": 0,
"call_id": "llm_fce8e965",
"turn": 1,
},
},
},
logging_obj=None,
)
assert event.type == "response.output_text.delta"
assert event.item_id == "msg_w_c4b77c96"
assert event.delta == "Hello there, friend!"
assert event.sequence_number == 12
assert event.content_index == 0
def test_reasoning_delta_becomes_a_reasoning_summary_delta(self):
"""`response.reasoning_summary_text.delta` is what LiteLLM already translates
into an Anthropic `thinking_delta`, which is the route Deep Research takes on
/v1/messages. It also keeps a separate item id from the answer text."""
config = _responses_config("apodex-1-1-deep-research")
event = config.transform_streaming_response(
model="apodex-1-1-deep-research",
parsed_chunk={
"type": "response.swarm.llm_delta",
"response_id": "w_c4b77c96",
"sequence_number": 7,
"swarm": {
"agent_id": "stateful_react",
"data": {"channel": "reasoning", "delta": "The user wants", "delta_index": 0},
},
},
logging_obj=None,
)
assert event.type == "response.reasoning_summary_text.delta"
assert event.item_id == "rs_w_c4b77c96"
assert event.delta == "The user wants"
assert event.summary_index == 0
assert not hasattr(event, "content_index")
def test_reasoning_and_answer_form_valid_anthropic_blocks(self):
config = _responses_config("apodex-1-1-deep-research")
reasoning_event = config.transform_streaming_response(
model="apodex-1-1-deep-research",
parsed_chunk={
"type": "response.swarm.llm_delta",
"response_id": "w_c4b77c96",
"sequence_number": 7,
"swarm": {"data": {"channel": "reasoning", "delta": "The user wants"}},
},
logging_obj=None,
)
answer_event = config.transform_streaming_response(
model="apodex-1-1-deep-research",
parsed_chunk={
"type": "response.swarm.llm_delta",
"response_id": "w_c4b77c96",
"sequence_number": 8,
"swarm": {"data": {"channel": "output_text", "delta": "Hello there, friend!"}},
},
logging_obj=None,
)
wrapper = AnthropicResponsesStreamWrapper(responses_stream=None, model="apodex-1-1-deep-research")
for event in (
reasoning_event,
answer_event,
{"type": "response.completed", "response": SimpleNamespace(status="completed", output=[], usage=None)},
):
wrapper._process_event(event)
chunks = list(wrapper._chunk_queue)
assert [(chunk["type"], chunk.get("index")) for chunk in chunks] == [
("content_block_start", 0),
("content_block_delta", 0),
("content_block_stop", 0),
("content_block_start", 1),
("content_block_delta", 1),
("content_block_stop", 1),
("message_delta", None),
("message_stop", None),
]
assert chunks[0]["content_block"]["type"] == "thinking"
assert chunks[1]["delta"] == {"type": "thinking_delta", "thinking": "The user wants"}
assert chunks[3]["content_block"]["type"] == "text"
assert chunks[4]["delta"] == {"type": "text_delta", "text": "Hello there, friend!"}
@pytest.mark.parametrize(
"channel",
(None, "tool_output", []),
ids=("no-channel", "unknown-channel", "non-string-channel"),
)
def test_intermediate_agent_deltas_are_not_claimed(self, channel):
"""The worker agent streams a draft answer on a channel-less delta.
Live capture: those four deltas spell "Hello, friend! How are you?" while the
reporter's `output_text` is the "Hello there, friend!" that lands in
response.completed. Claiming them would splice the draft into the answer.
"""
data = {"delta": "Hello,"} if channel is None else {"channel": channel, "delta": "Hello,"}
config = _responses_config("apodex-1-1-deep-research")
event = config.transform_streaming_response(
model="apodex-1-1-deep-research",
parsed_chunk={
"type": "response.swarm.llm_delta",
"response_id": "w_c4b77c96",
"sequence_number": 7,
"swarm": {"agent_id": "stateful_react", "data": data},
},
logging_obj=None,
)
assert event.type == "response.swarm.llm_delta"
def test_non_string_delta_is_not_claimed(self):
config = _responses_config("apodex-1-1-deep-research")
assert (
config._map_swarm_delta(
{
"type": "response.swarm.llm_delta",
"swarm": {"data": {"channel": "output_text", "delta": None}},
}
)
is None
)
@pytest.mark.parametrize(
"event_type",
(
"response.swarm.run_started",
"response.swarm.run_finished",
"response.swarm.injection_window",
"response.swarm.llm_attempt_started",
"response.swarm.llm_attempt_finished",
),
)
def test_swarm_lifecycle_events_pass_through(self, event_type: str):
"""LiteLLM cannot drop a chunk from the stream, so these stay as GenericEvent
rather than being silently swallowed; `run_finished` carries the final content."""
config = _responses_config("apodex-1-1-deep-research")
event = config.transform_streaming_response(
model="apodex-1-1-deep-research",
parsed_chunk={
"type": event_type,
"response_id": "w_c4b77c96",
"sequence_number": 4,
"swarm": {"agent_id": "reporter", "data": {"status": "success"}},
},
logging_obj=None,
)
assert event.type == event_type
def test_documented_events_pass_through_untouched(self):
config = _responses_config("apodex-1-1-deep-research")
event = config.transform_streaming_response(
model="apodex-1-1-deep-research",
parsed_chunk={"type": "response.in_progress", "sequence_number": 2},
logging_obj=None,
)
assert event.type == "response.in_progress"
def test_non_json_cancel_body_raises_the_provider_error(self):
"""The gateway answers a timed-out cancel with an HTML 504, not the JSON envelope."""
config = _responses_config("apodex-1-1-deep-research")
with pytest.raises(Exception, match="gateway timeout"):
config.transform_cancel_response_api_response(
raw_response=httpx.Response(504, content=b"<html>gateway timeout</html>"),
logging_obj=None,
)