refactor(apodex): move to a Python provider with model-aware transformations

The JSON provider path applies one contract to a whole provider, which is wrong
for Apodex: its two model families take different parameters. Replaces the
providers.json entry with litellm/llms/apodex/, reverting the shared
openai_like machinery to its original state.

/v1/responses is now keyed off the model. Core models are a stateless subset,
so store is pinned false and previous_response_id / background are rejected
rather than passed upstream to fail with a 400. The deep research tiers keep
all three, so background survives a client disconnect.

/v1/messages resolves per model too. Apodex serves the protocol natively for
the core models only, so the deep research tiers get no native config and fall
back to translation instead of hitting a path that does not serve them.

Chat completions pin stream to false for both families, drop tool params on the
deep research tiers, and rename max_completion_tokens to max_tokens. The
responses config also stops inheriting OpenAI's OPENAI_API_KEY fallback, which
would otherwise forward an unrelated OpenAI key to Apodex.

Tests live under tests/test_litellm/llms/apodex/ and touch no existing test file.
This commit is contained in:
zhanghanduo 2026-08-16 12:37:03 +08:00
parent 20ef10e920
commit 3b4ff9127b
20 changed files with 1001 additions and 438 deletions

View file

@ -99,7 +99,7 @@
"limit": 0
},
"reportUnknownArgumentType": {
"limit": 44774
"limit": 44776
},
"reportUnknownLambdaType": {
"limit": 113
@ -111,7 +111,7 @@
"limit": 19967
},
"reportUnknownVariableType": {
"limit": 30879
"limit": 30881
},
"reportUnnecessaryCast": {
"limit": 117

View file

@ -638,6 +638,7 @@ snowflake_models: Set = set()
gradient_ai_models: Set = set()
llama_models: Set = set()
nscale_models: Set = set()
apodex_models: Set = set()
nebius_models: Set = set()
nebius_embedding_models: Set = set()
aiml_models: Set = set()
@ -828,6 +829,8 @@ def _populate_provider_model_sets(model_cost_map: Dict) -> None:
llama_models.add(key)
elif value.get("litellm_provider") == "nscale":
nscale_models.add(key)
elif value.get("litellm_provider") == "apodex":
apodex_models.add(key)
elif value.get("litellm_provider") == "azure_ai":
azure_ai_models.add(key)
elif value.get("litellm_provider") == "voyage":
@ -1052,6 +1055,7 @@ model_list = list(
| llama_models
| featherless_ai_models
| nscale_models
| apodex_models
| deepgram_models
| elevenlabs_models
| dashscope_models
@ -1156,6 +1160,7 @@ def _build_models_by_provider() -> dict:
"gradient_ai": gradient_ai_models,
"meta_llama": llama_models,
"nscale": nscale_models,
"apodex": apodex_models,
"featherless_ai": featherless_ai_models,
"deepgram": deepgram_models,
"elevenlabs": elevenlabs_models,
@ -1782,6 +1787,9 @@ if TYPE_CHECKING:
from .llms.perplexity.responses.transformation import (
PerplexityResponsesConfig as PerplexityResponsesConfig,
)
from .llms.apodex.responses.transformation import (
ApodexResponsesConfig as ApodexResponsesConfig,
)
from .llms.databricks.responses.transformation import (
DatabricksResponsesAPIConfig as DatabricksResponsesAPIConfig,
)
@ -1855,6 +1863,7 @@ if TYPE_CHECKING:
PerplexityChatConfig as _PerplexityChatConfig,
)
from .llms.nscale.chat.transformation import NscaleConfig as _NscaleConfig
from .llms.apodex.chat.transformation import ApodexChatConfig as _ApodexChatConfig
from .llms.watsonx.chat.transformation import (
IBMWatsonXChatConfig as _IBMWatsonXChatConfig,
)
@ -1890,6 +1899,7 @@ if TYPE_CHECKING:
AzureOpenAIO1Config: Type[_AzureOpenAIO1Config]
PerplexityChatConfig: Type[_PerplexityChatConfig]
NscaleConfig: Type[_NscaleConfig]
ApodexChatConfig: Type[_ApodexChatConfig]
IBMWatsonXChatConfig: Type[_IBMWatsonXChatConfig]
IBMWatsonXAIConfig: Type[_IBMWatsonXAIConfig]
LiteLLMProxyChatConfig: Type[_LiteLLMProxyChatConfig]

View file

@ -238,6 +238,7 @@ LLM_CONFIG_NAMES: Final = (
"HostedVLLMResponsesAPIConfig",
"VolcEngineResponsesAPIConfig",
"PerplexityResponsesConfig",
"ApodexResponsesConfig",
"DatabricksResponsesAPIConfig",
"OpenRouterResponsesAPIConfig",
"BedrockMantleResponsesAPIConfig",
@ -291,6 +292,7 @@ LLM_CONFIG_NAMES: Final = (
"LmStudioEmbeddingConfig",
"NscaleConfig",
"PerplexityChatConfig",
"ApodexChatConfig",
"AzureOpenAIO1Config",
"IBMWatsonXAIConfig",
"IBMWatsonXChatConfig",
@ -961,6 +963,10 @@ _LLM_CONFIGS_IMPORT_MAP: Final = {
".llms.perplexity.responses.transformation",
"PerplexityResponsesConfig",
),
"ApodexResponsesConfig": (
".llms.apodex.responses.transformation",
"ApodexResponsesConfig",
),
"DatabricksResponsesAPIConfig": (
".llms.databricks.responses.transformation",
"DatabricksResponsesAPIConfig",
@ -1110,6 +1116,7 @@ _LLM_CONFIGS_IMPORT_MAP: Final = {
".llms.perplexity.chat.transformation",
"PerplexityChatConfig",
),
"ApodexChatConfig": (".llms.apodex.chat.transformation", "ApodexChatConfig"),
"AzureOpenAIO1Config": (
".llms.azure.chat.o_series_transformation",
"AzureOpenAIO1Config",

View file

@ -824,7 +824,7 @@ openai_compatible_providers: Final[list] = [
"pinstripes", # Pinstripes - JSON-configured provider
"darkbloom",
"meta", # Meta Model API (Muse Spark) - JSON-configured provider
"apodex", # Apodex - JSON-configured provider
"apodex",
]
openai_text_completion_compatible_providers: Final[list] = [ # providers that support `/v1/completions`
"together_ai",

View file

@ -349,9 +349,10 @@ def get_llm_provider(
elif endpoint == "https://api.meta.ai/v1":
custom_llm_provider = "meta"
dynamic_api_key = get_secret_str("META_API_KEY")
elif endpoint == "https://api.apodex.ai/v1":
custom_llm_provider = "apodex"
dynamic_api_key = get_secret_str("APODEX_API_KEY")
elif endpoint == litellm.ApodexChatConfig.API_BASE_URL:
custom_llm_provider = "apodex" # rebind-ok: dispatch chain resolves in place
# rebind-ok: dispatch chain resolves in place
dynamic_api_key = litellm.ApodexChatConfig.get_api_key()
if api_base is not None and not isinstance(api_base, str):
raise Exception(f"api base needs to be a string. api_base={api_base}")
@ -757,6 +758,9 @@ def _get_openai_compatible_provider_info(
api_base,
dynamic_api_key,
) = litellm.NscaleConfig()._get_openai_compatible_provider_info(api_base=api_base, api_key=api_key)
elif custom_llm_provider == "apodex":
api_base = litellm.ApodexChatConfig.get_api_base(api_base) # rebind-ok: dispatch chain resolves in place
dynamic_api_key = litellm.ApodexChatConfig.get_api_key(api_key) # rebind-ok: resolved in place
elif custom_llm_provider == "heroku":
(
api_base,

View file

@ -0,0 +1,119 @@
"""
Apodex chat completions — OpenAI-compatible, with two provider quirks:
- `stream` defaults to true upstream, so a non-streaming call has to say so
explicitly or Apodex answers with SSE that a plain call cannot parse
- the Deep Research tiers ignore sampling parameters and reject OpenAI-style
tools; only the core models take them
Ref: https://platform.apodex.ai/docs/chat-completions
"""
from collections.abc import Mapping
from typing import Final
from litellm.llms.openai.chat.gpt_transformation import OpenAIGPTConfig
from ..common_utils import (
APODEX_API_BASE_URL,
get_apodex_api_base,
get_apodex_api_key,
is_deep_research_model,
)
_DEEP_RESEARCH_PARAMS: Final = (
"max_tokens",
"max_completion_tokens",
"stream",
"stream_options",
"extra_headers",
"max_retries",
)
_CORE_PARAMS: Final = (
*_DEEP_RESEARCH_PARAMS,
"temperature",
"top_p",
"stop",
"seed",
"n",
"tools",
"tool_choice",
"function_call",
"functions",
"parallel_tool_calls",
)
class ApodexChatConfig(OpenAIGPTConfig):
"""
Reference: https://platform.apodex.ai/docs
API Key: APODEX_API_KEY
Default API Base: https://api.apodex.ai/v1
"""
API_BASE_URL = APODEX_API_BASE_URL
@property
def custom_llm_provider(self) -> str | None:
return "apodex"
@staticmethod
def get_api_key(api_key: str | None = None) -> str | None:
return get_apodex_api_key(api_key)
@staticmethod
def get_api_base(api_base: str | None = None) -> str | None:
return get_apodex_api_base(api_base)
def _get_openai_compatible_provider_info(
self, api_base: str | None, api_key: str | None
) -> tuple[str | None, str | None]:
return get_apodex_api_base(api_base), get_apodex_api_key(api_key)
def get_supported_openai_params(self, model: str) -> list: # mutable-ok: matches the base-class signature
supported: Final = _DEEP_RESEARCH_PARAMS if is_deep_research_model(model) else _CORE_PARAMS
return list(supported) # mutable-ok: matches the base-class signature
def map_openai_params(
self,
non_default_params: dict, # mutable-ok: matches the base-class signature
optional_params: dict, # mutable-ok: matches the base-class signature
model: str,
drop_params: bool,
) -> dict: # mutable-ok: matches the base-class signature
mapped: Final = super().map_openai_params(
non_default_params=non_default_params,
optional_params=optional_params,
model=model,
drop_params=drop_params,
)
# Apodex documents max_tokens only.
renamed: Final = (
mapped
if "max_completion_tokens" not in mapped
else { # mutable-ok: JSON request body
**{ # mutable-ok: JSON request body
key: value for key, value in mapped.items() if key != "max_completion_tokens"
},
"max_tokens": mapped["max_completion_tokens"],
}
)
if renamed.get("stream"):
return renamed
# The OpenAI SDK drops `stream` from the body when it is false, which would
# leave Apodex on its streaming default. extra_body is merged into the
# request body by the SDK, so it survives that drop.
requested_extra_body: Final = renamed.get("extra_body")
extra_body: Final = (
requested_extra_body
if isinstance(requested_extra_body, Mapping)
else {} # mutable-ok: JSON request body
)
return { # mutable-ok: JSON request body
**renamed,
"extra_body": {"stream": False, **extra_body}, # mutable-ok: JSON request body
}

View file

@ -0,0 +1,36 @@
"""
Shared helpers for the Apodex provider.
Apodex serves two model families on one base URL, and the model id picks which
contract applies. Core models (apodex-1.1, apodex-1.1-mini) are plain inference
with native sampling parameters. The Deep Research tiers run an agent that
plans, searches and iterates, so they ignore sampling parameters, reject
OpenAI-style tools, and keep server-side state.
Ref: https://platform.apodex.ai/docs/models
"""
from typing import Final
from litellm.secret_managers.main import get_secret_str
APODEX_API_BASE_URL: Final = "https://api.apodex.ai/v1"
_DEEP_RESEARCH_MARKER: Final = "-deep-"
def strip_provider_prefix(model: str) -> str:
return model.rpartition("/")[2]
def is_deep_research_model(model: str) -> bool:
"""True for the Deep Research / Solve / Discover tiers, e.g. apodex-1-1-deep-solve."""
return _DEEP_RESEARCH_MARKER in strip_provider_prefix(model)
def get_apodex_api_key(api_key: str | None = None) -> str | None:
return api_key or get_secret_str("APODEX_API_KEY")
def get_apodex_api_base(api_base: str | None = None) -> str:
return api_base or get_secret_str("APODEX_API_BASE") or APODEX_API_BASE_URL

View file

@ -0,0 +1,52 @@
"""
Apodex Anthropic Messages — native passthrough for the core models only.
Apodex implements the Anthropic protocol itself at POST /v1/messages and serves
the core models there, so the payload is forwarded untranslated and
Anthropic-only features such as `thinking` and `cache_control` survive. The Deep
Research tiers are not served on that path, so `ProviderConfigManager` hands back
no config for them and they fall back to LiteLLM's Anthropic-to-chat-completions
translation.
Ref: https://platform.apodex.ai/docs/anthropic-messages
"""
from litellm.llms.openai_like.messages.transformation import (
OpenAILikeAnthropicMessagesConfig,
)
from ..common_utils import get_apodex_api_base, get_apodex_api_key
class ApodexAnthropicMessagesConfig(OpenAILikeAnthropicMessagesConfig):
@property
def custom_llm_provider(self) -> str | None:
return "apodex"
def should_strip_billing_metadata(self) -> bool:
return True
def validate_anthropic_messages_environment(
self,
headers: dict[str, str], # mutable-ok: matches the base-class signature
model: str,
messages: list[object], # mutable-ok: matches the base-class signature
optional_params: dict, # mutable-ok: matches the base-class signature
litellm_params: dict, # mutable-ok: matches the base-class signature
api_key: str | None = None,
api_base: str | None = None,
) -> tuple[dict[str, str], str | None]: # mutable-ok: matches the base-class signature
"""Fill in the Apodex credentials and base URL.
The returned api_base is what the handler hands to get_complete_url, so
resolving it here is enough to reach the native endpoint.
"""
return super().validate_anthropic_messages_environment(
headers=headers,
model=model,
messages=messages,
optional_params=optional_params,
litellm_params=litellm_params,
api_key=get_apodex_api_key(api_key),
api_base=get_apodex_api_base(api_base),
)

View file

@ -0,0 +1,100 @@
"""
Apodex Responses API — OpenAI-compatible, with a model-aware parameter contract.
Apodex serves /v1/responses for both model families but they accept different
subsets, so the restrictions here are keyed off the model rather than applied
provider-wide:
- core models are a stateless subset: `store` is forced to false, and
`previous_response_id` or `background` come back as HTTP 400
- the Deep Research tiers keep server-side state, so they take all three
- both default `stream` to true, so a non-streaming call has to say so
Ref: https://platform.apodex.ai/docs/responses-api
https://platform.apodex.ai/docs/models
"""
from collections.abc import Mapping
from typing import Final
import litellm
from litellm.llms.openai.responses.transformation import OpenAIResponsesAPIConfig
from litellm.types.llms.openai import ResponsesAPIOptionalRequestParams
from litellm.types.router import GenericLiteLLMParams
from litellm.types.utils import LlmProviders
from ..common_utils import get_apodex_api_key, is_deep_research_model
# Rejected by the core models with HTTP 400: there is no server-side conversation
# to resume and requests are always executed inline.
_STATEFUL_PARAMS: Final = ("previous_response_id", "background")
class ApodexResponsesConfig(OpenAIResponsesAPIConfig):
@property
def custom_llm_provider(self) -> LlmProviders:
return LlmProviders.APODEX
def validate_environment(
self,
headers: dict, # mutable-ok: matches the base-class signature
model: str,
litellm_params: GenericLiteLLMParams | None,
) -> dict: # mutable-ok: matches the base-class signature
"""Resolve the Apodex key rather than inheriting OpenAI's OPENAI_API_KEY fallback,
which would otherwise forward an unrelated OpenAI key to Apodex."""
resolved_params: Final = litellm_params or GenericLiteLLMParams()
api_key: Final = get_apodex_api_key(resolved_params.api_key)
if api_key is None:
return headers
return { # mutable-ok: matches the base-class signature
**headers,
"Content-Type": "application/json",
"Authorization": f"Bearer {api_key}",
}
def get_supported_openai_params(self, model: str) -> list: # mutable-ok: matches the base-class signature
inherited: Final = super().get_supported_openai_params(model)
if is_deep_research_model(model):
return inherited
return [ # mutable-ok: matches the base-class signature
param for param in inherited if param not in _STATEFUL_PARAMS
]
def map_openai_params(
self,
response_api_optional_params: ResponsesAPIOptionalRequestParams,
model: str,
drop_params: bool,
) -> dict: # mutable-ok: matches the base-class signature
mapped: Final = super().map_openai_params(
response_api_optional_params=response_api_optional_params,
model=model,
drop_params=drop_params,
)
stateless: Final = (
mapped
if is_deep_research_model(model)
else self._enforce_stateless(mapped, model=model, drop_params=drop_params)
)
if stateless.get("stream"):
return {**stateless} # mutable-ok: JSON request body
return {**stateless, "stream": False} # mutable-ok: JSON request body
@staticmethod
def _enforce_stateless(params: Mapping[str, object], model: str, drop_params: bool) -> Mapping[str, object]:
"""Core models only: drop what the stateless subset rejects and pin store to false."""
if params.get("store") is True and not (drop_params or litellm.drop_params):
raise litellm.UnsupportedParamsError(
message=(
f"apodex model {model} does not support store=True on /v1/responses: the endpoint is a "
"stateless subset. To drop this, set `litellm.drop_params = True`"
),
status_code=400,
)
kept: Final = { # mutable-ok: JSON request body
key: value for key, value in params.items() if key not in _STATEFUL_PARAMS
}
return {**kept, "store": False} # mutable-ok: JSON request body

View file

@ -59,15 +59,7 @@ That's it! The provider will be automatically loaded and available.
// Optional: Special handling flags
"special_handling": {
"convert_content_list_to_string": true,
// Send "stream": false explicitly instead of omitting it. Needed by
// providers whose /v1/chat/completions and /v1/responses default to
// streaming, where omitting the field returns SSE to a non-streaming call
"send_explicit_stream_false": true,
// Always send "store": false on /v1/responses
"force_store_false": true
"convert_content_list_to_string": true
}
}
}

View file

@ -2,7 +2,7 @@
Dynamic configuration class generator for JSON-based providers.
"""
from collections.abc import Coroutine, Mapping
from collections.abc import Coroutine
from typing import Any, Final, Literal, overload
from litellm._logging import verbose_logger
@ -17,17 +17,6 @@ from litellm.types.llms.openai import AllMessageValues
from .json_loader import SimpleProviderConfig
def _clamp_temperature(temperature: float, n: int, constraints: Mapping[str, float]) -> float:
capped: Final = (
min(temperature, constraints["temperature_max"]) if "temperature_max" in constraints else temperature
)
floored: Final = max(capped, constraints["temperature_min"]) if "temperature_min" in constraints else capped
floor_for_multiple_choices: Final = constraints.get("temperature_min_with_n_gt_1")
if n > 1 and floor_for_multiple_choices is not None:
return max(floored, floor_for_multiple_choices)
return floored
def create_config_class(provider: SimpleProviderConfig):
"""Generate config class dynamically from JSON configuration"""
@ -142,36 +131,37 @@ def create_config_class(provider: SimpleProviderConfig):
"""Apply parameter mappings and constraints"""
supported_params: Final = self.get_supported_openai_params(model)
mapped: Final = {
**optional_params,
**{
provider.param_mappings.get(param, param): value
for param, value in non_default_params.items()
if param in provider.param_mappings or param in supported_params
},
}
constrained: Final = (
mapped
if "temperature" not in mapped
else {
**mapped,
"temperature": _clamp_temperature(
temperature=mapped["temperature"],
n=mapped.get("n", 1),
constraints=provider.constraints,
),
}
)
# Apply supported params
for param, value in non_default_params.items():
# Check parameter mappings first
if param in provider.param_mappings:
optional_params[provider.param_mappings[param]] = value
elif param in supported_params:
optional_params[param] = value
# The OpenAI SDK omits `stream` entirely when it is false, which makes
# stream-by-default providers answer a non-streaming call with SSE. Pin it
# on the wire through extra_body, which the SDK merges into the request body.
if not provider.special_handling.get("send_explicit_stream_false") or constrained.get("stream"):
return constrained
requested_extra_body: Final = constrained.get("extra_body")
extra_body: Final[dict] = requested_extra_body if isinstance(requested_extra_body, dict) else {}
return {**constrained, "extra_body": {"stream": False, **extra_body}}
# Apply temperature constraints if present
if "temperature" in optional_params:
temp = optional_params["temperature"]
constraints: Final = provider.constraints
# Clamp to max
if "temperature_max" in constraints:
temp = min(temp, constraints["temperature_max"])
# Clamp to min
if "temperature_min" in constraints:
temp = max(temp, constraints["temperature_min"])
# Special case: temperature_min_with_n_gt_1
if "temperature_min_with_n_gt_1" in constraints:
n: Final = optional_params.get("n", 1)
if n > 1 and temp < constraints["temperature_min_with_n_gt_1"]:
temp = constraints["temperature_min_with_n_gt_1"]
optional_params["temperature"] = temp
return optional_params
@property
def custom_llm_provider(self) -> str | None:
@ -242,8 +232,6 @@ def create_responses_config_class(provider: SimpleProviderConfig):
) -> dict:
if provider.special_handling.get("force_store_false"):
response_api_optional_request_params["store"] = False
if provider.special_handling.get("send_explicit_stream_false"):
response_api_optional_request_params.setdefault("stream", False)
return super().transform_responses_api_request(
model=model,
input=input,

View file

@ -175,18 +175,6 @@
"base_class": "openai_gpt",
"supported_endpoints": ["/v1/chat/completions", "/v1/responses", "/v1/messages"]
},
"apodex": {
"base_url": "https://api.apodex.ai/v1",
"api_key_env": "APODEX_API_KEY",
"api_base_env": "APODEX_API_BASE",
"param_mappings": {
"max_completion_tokens": "max_tokens"
},
"supported_endpoints": ["/v1/chat/completions", "/v1/responses", "/v1/messages"],
"special_handling": {
"send_explicit_stream_false": true
}
},
"pinstripes": {
"base_url": "https://pinstripes.io/v1",
"api_key_env": "PINSTRIPES_API_KEY",

View file

@ -7928,6 +7928,7 @@ class ProviderConfigManager:
),
LlmProviders.GRADIENT_AI: (lambda: litellm.GradientAIConfig(), False),
LlmProviders.NSCALE: (lambda: litellm.NscaleConfig(), False),
LlmProviders.APODEX: (lambda: litellm.ApodexChatConfig(), False),
LlmProviders.HEROKU: (lambda: litellm.HerokuChatConfig(), False),
LlmProviders.OCI: (lambda: litellm.OCIChatConfig(), False),
LlmProviders.HYPERBOLIC: (lambda: litellm.HyperbolicChatConfig(), False),
@ -8255,6 +8256,17 @@ class ProviderConfigManager:
)
return DeepSeekAnthropicMessagesConfig()
elif litellm.LlmProviders.APODEX == provider:
from litellm.llms.apodex.common_utils import is_deep_research_model
from litellm.llms.apodex.messages.transformation import (
ApodexAnthropicMessagesConfig,
)
# Apodex only serves the core models on its native /v1/messages path; the
# deep research tiers get no config so they fall back to translation.
if is_deep_research_model(model):
return None
return ApodexAnthropicMessagesConfig()
elif litellm.LlmProviders.TENCENT == provider:
from litellm.llms.tencent.messages.transformation import (
TencentAnthropicMessagesConfig,
@ -8440,6 +8452,8 @@ class ProviderConfigManager:
return litellm.ManusResponsesAPIConfig()
elif litellm.LlmProviders.PERPLEXITY == provider:
return litellm.PerplexityResponsesConfig()
elif litellm.LlmProviders.APODEX == provider:
return litellm.ApodexResponsesConfig()
elif litellm.LlmProviders.DATABRICKS == provider:
# Databricks Responses API is only compatible with OpenAI GPT models
if model and "gpt" in model.lower():

View file

@ -0,0 +1,209 @@
"""
Apodex chat completions transformation.
"""
import json
import httpx
import openai
import pytest
import litellm
from litellm.types.utils import LlmProviders
from litellm.utils import ProviderConfigManager
CORE_MODEL = "apodex/apodex-1.1"
DEEP_RESEARCH_MODEL = "apodex/apodex-1-1-deep-research"
CHAT_RESPONSE = {
"id": "chatcmpl-abc123",
"object": "chat.completion",
"created": 1712345678,
"model": "apodex-1.1",
"choices": [
{
"index": 0,
"message": {"role": "assistant", "content": "ok", "reasoning_content": "let me think"},
"finish_reason": "stop",
}
],
"usage": {
"prompt_tokens": 1000,
"completion_tokens": 100,
"total_tokens": 1100,
"prompt_tokens_details": {"cached_tokens": 500},
},
}
STREAM_BODY = (
b'data: {"id":"chatcmpl-abc123","object":"chat.completion.chunk","created":1,"model":"apodex-1.1",'
b'"choices":[{"index":0,"delta":{"content":"ok"},"finish_reason":null}]}\n\n'
b"data: [DONE]\n\n"
)
@pytest.fixture(autouse=True)
def _apodex_env(monkeypatch: pytest.MonkeyPatch):
"""Resolve models against the in-repo cost map, not the published one."""
monkeypatch.setenv("APODEX_API_KEY", "sk-apodex-test")
monkeypatch.delenv("APODEX_API_BASE", raising=False)
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
yield
def _client(captured: dict, *, stream: bool = False) -> openai.OpenAI:
def handler(request: httpx.Request) -> httpx.Response:
captured["url"] = str(request.url)
captured["body"] = json.loads(request.content)
if stream:
return httpx.Response(200, headers={"content-type": "text/event-stream"}, content=STREAM_BODY)
return httpx.Response(200, json=CHAT_RESPONSE)
return openai.OpenAI(
api_key="sk-apodex-test",
base_url="https://api.apodex.ai/v1",
http_client=httpx.Client(transport=httpx.MockTransport(handler)),
)
def _chat_config(model: str):
return ProviderConfigManager.get_provider_chat_config(model=model, provider=LlmProviders.APODEX)
class TestProviderResolution:
def test_prefixed_model_resolves_to_the_default_base(self):
model, provider, api_key, api_base = litellm.get_llm_provider(model=CORE_MODEL)
assert (model, provider, api_key, api_base) == (
"apodex-1.1",
"apodex",
"sk-apodex-test",
"https://api.apodex.ai/v1",
)
def test_api_base_autodetection(self):
_, provider, api_key, _ = litellm.get_llm_provider(model="apodex-1.1", api_base="https://api.apodex.ai/v1")
assert provider == "apodex"
assert api_key == "sk-apodex-test"
def test_explicit_api_base_and_key_win(self):
_, provider, api_key, api_base = litellm.get_llm_provider(
model=CORE_MODEL, api_base="https://gateway.internal/v1", api_key="sk-override"
)
assert (provider, api_key, api_base) == ("apodex", "sk-override", "https://gateway.internal/v1")
def test_api_base_env_override(self, monkeypatch: pytest.MonkeyPatch):
monkeypatch.setenv("APODEX_API_BASE", "https://env.apodex.test/v1")
_, _, _, api_base = litellm.get_llm_provider(model=CORE_MODEL)
assert api_base == "https://env.apodex.test/v1"
class TestStreamDefault:
"""Apodex defaults `stream` to true, so a non-streaming call has to pin it to false.
Regression guard: the OpenAI SDK drops `stream` from the body when it is false,
which would leave Apodex streaming SSE at a call that cannot parse it.
"""
def test_non_streaming_call_pins_stream_false(self):
captured: dict = {}
response = litellm.completion(
model=CORE_MODEL,
messages=[{"role": "user", "content": "hi"}],
client=_client(captured),
)
assert captured["url"] == "https://api.apodex.ai/v1/chat/completions"
assert captured["body"]["stream"] is False
assert captured["body"]["model"] == "apodex-1.1"
assert response.choices[0].message.reasoning_content == "let me think"
def test_streaming_call_sends_stream_true(self):
captured: dict = {}
chunks = list(
litellm.completion(
model=CORE_MODEL,
messages=[{"role": "user", "content": "hi"}],
stream=True,
client=_client(captured, stream=True),
)
)
assert captured["body"]["stream"] is True
assert chunks
def test_deep_research_models_pin_stream_too(self):
captured: dict = {}
litellm.completion(
model=DEEP_RESEARCH_MODEL,
messages=[{"role": "user", "content": "hi"}],
client=_client(captured),
)
assert captured["body"]["stream"] is False
def test_user_supplied_extra_body_is_preserved(self):
"""Deep research tiers reach external tools through `mcp_servers` in extra_body."""
captured: dict = {}
mcp_servers = [{"name": "docs", "url": "https://example.com/mcp"}]
litellm.completion(
model=DEEP_RESEARCH_MODEL,
messages=[{"role": "user", "content": "hi"}],
extra_body={"mcp_servers": mcp_servers},
client=_client(captured),
)
assert captured["body"]["stream"] is False
assert captured["body"]["mcp_servers"] == mcp_servers
class TestSupportedParams:
def test_core_models_support_tools(self):
supported = _chat_config("apodex-1.1").get_supported_openai_params("apodex-1.1")
assert "tools" in supported
assert "tool_choice" in supported
assert "temperature" in supported
assert "top_p" in supported
def test_deep_research_rejects_tools_and_sampling_params(self):
"""The tiers document tools as unsupported and sampling params as ignored."""
supported = _chat_config(DEEP_RESEARCH_MODEL).get_supported_openai_params("apodex-1-1-deep-research")
for param in ("tools", "tool_choice", "function_call", "functions", "parallel_tool_calls"):
assert param not in supported
assert "temperature" not in supported
assert "top_p" not in supported
assert "max_tokens" in supported
def test_tools_on_a_deep_research_model_raise(self):
with pytest.raises(litellm.UnsupportedParamsError, match="tools"):
litellm.completion(
model=DEEP_RESEARCH_MODEL,
messages=[{"role": "user", "content": "hi"}],
tools=[{"type": "function", "function": {"name": "f", "parameters": {}}}],
client=_client({}),
)
def test_max_completion_tokens_is_renamed_to_max_tokens(self):
captured: dict = {}
litellm.completion(
model=CORE_MODEL,
messages=[{"role": "user", "content": "hi"}],
max_completion_tokens=512,
client=_client(captured),
)
assert captured["body"]["max_tokens"] == 512
assert "max_completion_tokens" not in captured["body"]
class TestCostTracking:
def test_cached_input_is_billed_at_the_lower_rate(self):
captured: dict = {}
response = litellm.completion(
model=CORE_MODEL,
messages=[{"role": "user", "content": "hi"}],
client=_client(captured),
)
# 500 fresh input + 500 cached input + 100 output
expected = 500 * 3e-07 + 500 * 3e-08 + 100 * 3e-06
assert litellm.completion_cost(response, model=CORE_MODEL) == pytest.approx(expected)

View file

@ -0,0 +1,136 @@
"""
Apodex provider registration and model-family classification.
"""
import json
from pathlib import Path
import pytest
import litellm
from litellm.llms.apodex.common_utils import (
APODEX_API_BASE_URL,
get_apodex_api_base,
get_apodex_api_key,
is_deep_research_model,
)
from litellm.types.utils import LlmProviders
REPO_ROOT = Path(__file__).parents[4]
CORE_MODELS = ("apodex-1.1", "apodex-1.1-mini")
DEEP_RESEARCH_MODELS = (
"apodex-1-1-deep-research",
"apodex-1-1-deep-solve",
"apodex-1-1-deep-discover",
"apodex-1-0-deep-research",
"apodex-1-0-deep-solve",
"apodex-1-0-deep-discover",
)
class TestModelFamily:
"""The model id, not the provider, selects which Apodex contract applies."""
@pytest.mark.parametrize("model", CORE_MODELS)
def test_core_models_are_not_deep_research(self, model: str):
assert is_deep_research_model(model) is False
assert is_deep_research_model(f"apodex/{model}") is False
@pytest.mark.parametrize("model", DEEP_RESEARCH_MODELS)
def test_deep_research_models_are_detected(self, model: str):
assert is_deep_research_model(model) is True
assert is_deep_research_model(f"apodex/{model}") is True
def test_prefix_does_not_leak_into_classification(self):
"""A provider prefix containing the marker must not flip a core model."""
assert is_deep_research_model("some-deep-gateway/apodex-1.1") is False
class TestCredentialResolution:
def test_defaults(self, monkeypatch: pytest.MonkeyPatch):
monkeypatch.delenv("APODEX_API_BASE", raising=False)
monkeypatch.setenv("APODEX_API_KEY", "sk-env")
assert get_apodex_api_base(None) == APODEX_API_BASE_URL
assert get_apodex_api_key(None) == "sk-env"
def test_explicit_values_win_over_env(self, monkeypatch: pytest.MonkeyPatch):
monkeypatch.setenv("APODEX_API_BASE", "https://env.apodex.test/v1")
monkeypatch.setenv("APODEX_API_KEY", "sk-env")
assert get_apodex_api_base("https://explicit.apodex.test/v1") == "https://explicit.apodex.test/v1"
assert get_apodex_api_key("sk-explicit") == "sk-explicit"
def test_env_base_overrides_default(self, monkeypatch: pytest.MonkeyPatch):
monkeypatch.setenv("APODEX_API_BASE", "https://env.apodex.test/v1")
assert get_apodex_api_base(None) == "https://env.apodex.test/v1"
class TestRegistration:
def test_provider_enum_and_lists(self):
assert LlmProviders.APODEX.value == "apodex"
assert "apodex" in litellm.provider_list
assert "apodex" in litellm.constants.openai_compatible_providers
assert APODEX_API_BASE_URL in litellm.constants.openai_compatible_endpoints
def test_not_registered_as_a_json_provider(self):
"""Apodex needs model-aware transformations, so it must not fall into the
generic JSON path, which would shadow the Python configs in provider resolution."""
from litellm.llms.openai_like.json_loader import JSONProviderRegistry
assert JSONProviderRegistry.exists("apodex") is False
def test_config_classes_resolve_from_the_lazy_registry(self):
assert litellm.ApodexChatConfig().custom_llm_provider == "apodex"
assert litellm.ApodexResponsesConfig().custom_llm_provider == LlmProviders.APODEX
class TestModelMetadata:
@pytest.fixture(scope="class")
def model_cost(self) -> dict:
with open(REPO_ROOT / "model_prices_and_context_window.json") as f:
return json.load(f)
def test_every_apodex_model_is_registered(self, model_cost: dict):
assert {key for key in model_cost if key.startswith("apodex/")} == {
f"apodex/{model}" for model in (*CORE_MODELS, *DEEP_RESEARCH_MODELS)
}
def test_core_model_pricing(self, model_cost: dict):
info = model_cost["apodex/apodex-1.1"]
assert info["litellm_provider"] == "apodex"
assert info["mode"] == "chat"
assert info["max_input_tokens"] == 262144
assert info["input_cost_per_token"] == 3e-07
assert info["cache_read_input_token_cost"] == 3e-08
assert info["output_cost_per_token"] == 3e-06
# Requests over 200K input tokens are billed at 2x across every tier
assert info["input_cost_per_token_above_200k_tokens"] == 6e-07
assert info["cache_read_input_token_cost_above_200k_tokens"] == 6e-08
assert info["output_cost_per_token_above_200k_tokens"] == 6e-06
assert info["supports_prompt_caching"] is True
assert info["supports_function_calling"] is True
assert info["supported_endpoints"] == ["/v1/chat/completions", "/v1/responses", "/v1/messages"]
def test_deep_research_model_pricing(self, model_cost: dict):
info = model_cost["apodex/apodex-1-1-deep-research"]
assert info["max_input_tokens"] == 131072
assert info["max_output_tokens"] == 65536
assert info["input_cost_per_token"] == 5e-06
assert info["output_cost_per_token"] == 2e-05
assert info["supports_function_calling"] is False
assert info["supports_response_schema"] is False
assert info["supports_prompt_caching"] is False
assert info["supports_web_search"] is True
@pytest.mark.parametrize("model", DEEP_RESEARCH_MODELS)
def test_deep_research_models_are_not_on_the_native_messages_path(self, model_cost: dict, model: str):
"""Apodex serves /v1/messages for the core models only."""
assert "/v1/messages" not in model_cost[f"apodex/{model}"]["supported_endpoints"]
def test_backup_cost_map_in_sync(self, model_cost: dict):
with open(REPO_ROOT / "litellm" / "model_prices_and_context_window_backup.json") as f:
backup = json.load(f)
for key in (key for key in model_cost if key.startswith("apodex/")):
assert backup[key] == model_cost[key], f"{key} differs between main and backup cost maps"

View file

@ -0,0 +1,98 @@
"""
Apodex Anthropic Messages transformation.
Apodex implements the Anthropic protocol natively at POST /v1/messages, but only
serves the core models there. The Deep Research tiers must keep working on the
same route through LiteLLM's translation instead of being handed to a path that
would reject them.
"""
import pytest
import litellm
from litellm.types.utils import LlmProviders
from litellm.utils import ProviderConfigManager
CORE_MODELS = ("apodex-1.1", "apodex-1.1-mini")
DEEP_RESEARCH_MODELS = (
"apodex-1-1-deep-research",
"apodex-1-1-deep-solve",
"apodex-1-1-deep-discover",
"apodex-1-0-deep-research",
"apodex-1-0-deep-solve",
"apodex-1-0-deep-discover",
)
@pytest.fixture(autouse=True)
def _apodex_env(monkeypatch: pytest.MonkeyPatch):
monkeypatch.setenv("APODEX_API_KEY", "sk-apodex-test")
monkeypatch.delenv("APODEX_API_BASE", raising=False)
yield
def _messages_config(model: str):
return ProviderConfigManager.get_provider_anthropic_messages_config(model=model, provider=LlmProviders.APODEX)
def _complete_url() -> str:
"""Resolve the endpoint the way the handler does: validate first, then build the URL.
validate_anthropic_messages_environment returns the api_base the handler feeds
into get_complete_url, so the two steps have to run in that order.
"""
config = _messages_config("apodex-1.1")
assert config is not None
_, api_base = config.validate_anthropic_messages_environment(
headers={}, model="apodex-1.1", messages=[], optional_params={}, litellm_params={}
)
return config.get_complete_url(
api_base=api_base, api_key=None, model="apodex-1.1", optional_params={}, litellm_params={}
)
class TestNativePassthroughRouting:
@pytest.mark.parametrize("model", CORE_MODELS)
def test_core_models_get_the_native_config(self, model: str):
config = _messages_config(model)
assert config is not None
assert type(config).__name__ == "ApodexAnthropicMessagesConfig"
assert config.custom_llm_provider == "apodex"
@pytest.mark.parametrize("model", DEEP_RESEARCH_MODELS)
def test_deep_research_models_fall_back_to_translation(self, model: str):
"""No native config means LiteLLM translates to chat completions, which works,
instead of forwarding to a path Apodex does not serve for these tiers."""
assert _messages_config(model) is None
class TestNativePassthroughRequest:
def test_url_targets_the_native_messages_path(self):
assert _complete_url() == "https://api.apodex.ai/v1/messages"
def test_url_honours_an_api_base_override(self, monkeypatch: pytest.MonkeyPatch):
monkeypatch.setenv("APODEX_API_BASE", "https://env.apodex.test/v1")
assert _complete_url() == "https://env.apodex.test/v1/messages"
def test_headers_use_the_provider_api_key(self):
config = _messages_config("apodex-1.1")
assert config is not None
headers, _ = config.validate_anthropic_messages_environment(
headers={}, model="apodex-1.1", messages=[], optional_params={}, litellm_params={}
)
assert headers["authorization"] == "Bearer sk-apodex-test"
assert headers["anthropic-version"] == "2023-06-01"
assert headers["content-type"] == "application/json"
def test_caller_supplied_auth_header_is_not_overwritten(self):
config = _messages_config("apodex-1.1")
assert config is not None
headers, _ = config.validate_anthropic_messages_environment(
headers={"x-api-key": "sk-caller"},
model="apodex-1.1",
messages=[],
optional_params={},
litellm_params={},
)
assert headers["x-api-key"] == "sk-caller"
assert "authorization" not in headers

View file

@ -0,0 +1,177 @@
"""
Apodex Responses API transformation.
The core models expose a stateless subset of /v1/responses while the Deep
Research tiers keep server-side state, so the parameter contract is keyed off
the model rather than applied provider-wide.
"""
import pytest
import litellm
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler
from litellm.types.utils import LlmProviders
from litellm.utils import ProviderConfigManager
CORE_MODEL = "apodex/apodex-1.1"
CORE_MINI_MODEL = "apodex/apodex-1.1-mini"
DEEP_RESEARCH_MODEL = "apodex/apodex-1-1-deep-research"
@pytest.fixture(autouse=True)
def _apodex_env(monkeypatch: pytest.MonkeyPatch):
monkeypatch.setenv("APODEX_API_KEY", "sk-apodex-test")
monkeypatch.delenv("APODEX_API_BASE", raising=False)
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
monkeypatch.setattr(litellm, "drop_params", False)
yield
_SENTINEL = "apodex-request-captured"
def _capture(**kwargs) -> dict:
"""Run litellm.responses() and return the request it would have sent.
Validation errors raised before the request is built propagate to the caller.
"""
captured: dict = {}
class CapturingHandler(HTTPHandler):
def post(self, *args, **post_kwargs):
captured.update(url=post_kwargs.get("url"), body=post_kwargs.get("json"))
raise RuntimeError(_SENTINEL)
try:
litellm.responses(client=CapturingHandler(), **kwargs)
except Exception as exc:
if _SENTINEL not in str(exc):
raise
assert captured, "no request was sent"
return captured
def _responses_config(model: str):
return ProviderConfigManager.get_provider_responses_api_config(model=model, provider=LlmProviders.APODEX)
class TestConfigSelection:
def test_python_config_is_used_for_every_apodex_model(self):
for model in ("apodex-1.1", "apodex-1.1-mini", "apodex-1-1-deep-research"):
config = _responses_config(model)
assert type(config).__name__ == "ApodexResponsesConfig"
def test_auth_uses_the_apodex_key(self):
config = _responses_config("apodex-1.1")
assert config.validate_environment(headers={}, model="apodex-1.1", litellm_params=None) == {
"Content-Type": "application/json",
"Authorization": "Bearer sk-apodex-test",
}
def test_auth_does_not_fall_back_to_an_openai_key(self, monkeypatch: pytest.MonkeyPatch):
"""The inherited OpenAI config would forward OPENAI_API_KEY to Apodex."""
monkeypatch.delenv("APODEX_API_KEY", raising=False)
monkeypatch.setenv("OPENAI_API_KEY", "sk-openai-must-not-leak")
config = _responses_config("apodex-1.1")
assert config.validate_environment(headers={}, model="apodex-1.1", litellm_params=None) == {}
def test_request_targets_the_apodex_responses_url(self):
assert _capture(model=CORE_MODEL, input="hi")["url"] == "https://api.apodex.ai/v1/responses"
def test_request_honours_an_api_base_override(self, monkeypatch: pytest.MonkeyPatch):
monkeypatch.setenv("APODEX_API_BASE", "https://env.apodex.test/v1")
assert _capture(model=CORE_MODEL, input="hi")["url"] == "https://env.apodex.test/v1/responses"
class TestStreamDefault:
def test_non_streaming_pins_stream_false(self):
captured = _capture(model=DEEP_RESEARCH_MODEL, input="hi")
assert captured["url"] == "https://api.apodex.ai/v1/responses"
assert captured["body"]["stream"] is False
@pytest.mark.asyncio
async def test_streaming_sends_stream_true(self):
captured: dict = {}
class CapturingHandler(AsyncHTTPHandler):
async def post(self, *args, **kwargs):
captured.update(body=kwargs.get("json"))
raise RuntimeError("captured")
with pytest.raises(Exception, match="captured"):
await litellm.aresponses(
model=DEEP_RESEARCH_MODEL, input="hi", stream=True, client=CapturingHandler()
)
assert captured["body"]["stream"] is True
class TestCoreModelStatelessSubset:
"""Apodex core models reject anything that would persist state on their side."""
@pytest.mark.parametrize("model", [CORE_MODEL, CORE_MINI_MODEL])
def test_store_is_pinned_false(self, model: str):
captured = _capture(model=model, input="hi")
assert captured["body"]["store"] is False
def test_store_true_raises(self):
with pytest.raises(litellm.UnsupportedParamsError, match="store=True"):
_capture(model=CORE_MODEL, input="hi", store=True)
def test_background_raises(self):
with pytest.raises(litellm.UnsupportedParamsError, match="background"):
_capture(model=CORE_MODEL, input="hi", background=True)
def test_previous_response_id_raises(self):
with pytest.raises(litellm.UnsupportedParamsError, match="previous_response_id"):
_capture(model=CORE_MODEL, input="hi", previous_response_id="resp_1")
def test_drop_params_strips_all_three(self, monkeypatch: pytest.MonkeyPatch):
monkeypatch.setattr(litellm, "drop_params", True)
captured = _capture(
model=CORE_MODEL,
input="hi",
store=True,
background=True,
previous_response_id="resp_1",
)
assert captured["body"]["store"] is False
assert "background" not in captured["body"]
assert "previous_response_id" not in captured["body"]
def test_stateful_params_are_not_advertised(self):
supported = _responses_config("apodex-1.1").get_supported_openai_params("apodex-1.1")
assert "background" not in supported
assert "previous_response_id" not in supported
assert "max_output_tokens" in supported
def test_max_output_tokens_still_passes_through(self):
captured = _capture(model=CORE_MODEL, input="hi", max_output_tokens=512)
assert captured["body"]["max_output_tokens"] == 512
class TestDeepResearchKeepsState:
"""The agent tiers survive client disconnects, so none of this may be stripped."""
def test_background_passes_through(self):
captured = _capture(model=DEEP_RESEARCH_MODEL, input="hi", background=True)
assert captured["body"]["background"] is True
def test_store_and_previous_response_id_pass_through(self):
captured = _capture(model=DEEP_RESEARCH_MODEL, input="hi", store=True, previous_response_id="resp_1")
assert captured["body"]["store"] is True
assert captured["body"]["previous_response_id"] == "resp_1"
def test_store_is_not_pinned_when_unset(self):
captured = _capture(model=DEEP_RESEARCH_MODEL, input="hi")
assert "store" not in captured["body"]
def test_stateful_params_are_advertised(self):
supported = _responses_config("apodex-1-1-deep-research").get_supported_openai_params(
"apodex-1-1-deep-research"
)
assert "background" in supported
assert "previous_response_id" in supported

View file

@ -1,310 +0,0 @@
"""
Tests for the Apodex provider (https://platform.apodex.ai/docs).
"""
import json
from pathlib import Path
import httpx
import openai
import pytest
import litellm
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler
from litellm.types.utils import LlmProviders
from litellm.utils import ProviderConfigManager
REPO_ROOT = Path(__file__).parents[4]
CORE_MODEL = "apodex/apodex-1.1"
DEEP_RESEARCH_MODEL = "apodex/apodex-1-1-deep-research"
CHAT_RESPONSE = {
"id": "chatcmpl-abc123",
"object": "chat.completion",
"created": 1712345678,
"model": "apodex-1.1",
"choices": [
{
"index": 0,
"message": {"role": "assistant", "content": "ok", "reasoning_content": "let me think"},
"finish_reason": "stop",
}
],
"usage": {
"prompt_tokens": 1000,
"completion_tokens": 100,
"total_tokens": 1100,
"prompt_tokens_details": {"cached_tokens": 500},
},
}
STREAM_BODY = (
b'data: {"id":"chatcmpl-abc123","object":"chat.completion.chunk","created":1,"model":"apodex-1.1",'
b'"choices":[{"index":0,"delta":{"content":"ok"},"finish_reason":null}]}\n\n'
b"data: [DONE]\n\n"
)
@pytest.fixture(autouse=True)
def _apodex_env(monkeypatch: pytest.MonkeyPatch):
"""Resolve models against the in-repo cost map, not the published one."""
monkeypatch.setenv("APODEX_API_KEY", "sk-apodex-test")
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
yield
def _openai_client(captured: dict, *, stream: bool = False) -> openai.OpenAI:
def handler(request: httpx.Request) -> httpx.Response:
captured["url"] = str(request.url)
captured["body"] = json.loads(request.content)
if stream:
return httpx.Response(200, headers={"content-type": "text/event-stream"}, content=STREAM_BODY)
return httpx.Response(200, json=CHAT_RESPONSE)
return openai.OpenAI(
api_key="sk-apodex-test",
base_url="https://api.apodex.ai/v1",
http_client=httpx.Client(transport=httpx.MockTransport(handler)),
)
class TestApodexRegistration:
def test_provider_enum_and_lists(self):
assert LlmProviders.APODEX.value == "apodex"
assert "apodex" in litellm.provider_list
assert "apodex" in litellm.constants.openai_compatible_providers
def test_json_provider_config(self):
from litellm.llms.openai_like.json_loader import JSONProviderRegistry
apodex = JSONProviderRegistry.get("apodex")
assert apodex is not None
assert apodex.base_url == "https://api.apodex.ai/v1"
assert apodex.api_key_env == "APODEX_API_KEY"
assert apodex.api_base_env == "APODEX_API_BASE"
assert apodex.param_mappings["max_completion_tokens"] == "max_tokens"
assert apodex.supported_endpoints == ["/v1/chat/completions", "/v1/responses", "/v1/messages"]
assert JSONProviderRegistry.supports_responses_api("apodex") is True
def test_provider_resolution(self):
model, provider, _, api_base = litellm.get_llm_provider(model=CORE_MODEL)
assert (model, provider, api_base) == ("apodex-1.1", "apodex", "https://api.apodex.ai/v1")
def test_api_base_autodetection(self):
_, provider, api_key, _ = litellm.get_llm_provider(model="apodex-1.1", api_base="https://api.apodex.ai/v1")
assert provider == "apodex"
assert api_key == "sk-apodex-test"
def test_explicit_api_base_and_key_win(self):
_, provider, api_key, api_base = litellm.get_llm_provider(
model=CORE_MODEL, api_base="https://gateway.internal/v1", api_key="sk-override"
)
assert (provider, api_key, api_base) == ("apodex", "sk-override", "https://gateway.internal/v1")
class TestApodexStreamDefault:
"""Apodex defaults `stream` to true, so a non-streaming call must pin it to false.
Regression guard: the OpenAI SDK omits `stream` when it is false, which would make
litellm.completion() receive SSE and fail to parse it.
"""
def test_chat_completion_pins_stream_false(self):
captured: dict = {}
response = litellm.completion(
model=CORE_MODEL,
messages=[{"role": "user", "content": "hi"}],
client=_openai_client(captured),
)
assert captured["url"] == "https://api.apodex.ai/v1/chat/completions"
assert captured["body"]["stream"] is False
assert captured["body"]["model"] == "apodex-1.1"
assert response.choices[0].message.reasoning_content == "let me think"
def test_chat_completion_streaming_sends_stream_true(self):
captured: dict = {}
chunks = list(
litellm.completion(
model=CORE_MODEL,
messages=[{"role": "user", "content": "hi"}],
stream=True,
client=_openai_client(captured, stream=True),
)
)
assert captured["body"]["stream"] is True
assert chunks
def test_user_supplied_extra_body_is_preserved(self):
captured: dict = {}
litellm.completion(
model=CORE_MODEL,
messages=[{"role": "user", "content": "hi"}],
extra_body={"mcp_servers": [{"name": "docs", "url": "https://example.com/mcp"}]},
client=_openai_client(captured),
)
assert captured["body"]["stream"] is False
assert captured["body"]["mcp_servers"] == [{"name": "docs", "url": "https://example.com/mcp"}]
def test_responses_api_pins_stream_false(self):
captured: dict = {}
class CapturingHandler(HTTPHandler):
def post(self, *args, **kwargs):
captured.update(url=kwargs.get("url"), body=kwargs.get("json"))
raise RuntimeError("captured")
with pytest.raises(Exception):
litellm.responses(model=DEEP_RESEARCH_MODEL, input="hi", client=CapturingHandler())
assert captured["url"] == "https://api.apodex.ai/v1/responses"
assert captured["body"]["stream"] is False
assert captured["body"]["model"] == "apodex-1-1-deep-research"
@pytest.mark.asyncio
async def test_responses_api_streaming_sends_stream_true(self):
captured: dict = {}
class CapturingHandler(AsyncHTTPHandler):
async def post(self, *args, **kwargs):
captured.update(body=kwargs.get("json"))
raise RuntimeError("captured")
with pytest.raises(Exception):
await litellm.aresponses(model=DEEP_RESEARCH_MODEL, input="hi", stream=True, client=CapturingHandler())
assert captured["body"]["stream"] is True
def test_flag_is_opt_in_for_other_json_providers(self):
from litellm.llms.openai_like.json_loader import JSONProviderRegistry
pinstripes = JSONProviderRegistry.get("pinstripes")
assert pinstripes is not None
assert "send_explicit_stream_false" not in pinstripes.special_handling
config = ProviderConfigManager.get_provider_chat_config(
model="ps/glm-4.5-air", provider=LlmProviders.PINSTRIPES
)
params = config.map_openai_params({}, {}, "ps/glm-4.5-air", False)
assert "stream" not in params
assert "stream" not in (params.get("extra_body") or {})
class TestApodexToolSupport:
"""Deep research tiers reject OpenAI-style tools; core models accept them."""
def test_deep_research_drops_tool_params(self):
config = ProviderConfigManager.get_provider_chat_config(
model="apodex-1-1-deep-research", provider=LlmProviders.APODEX
)
supported = config.get_supported_openai_params("apodex-1-1-deep-research")
assert "tools" not in supported
assert "tool_choice" not in supported
def test_core_model_keeps_tool_params(self):
config = ProviderConfigManager.get_provider_chat_config(model="apodex-1.1", provider=LlmProviders.APODEX)
supported = config.get_supported_openai_params("apodex-1.1")
assert "tools" in supported
assert "tool_choice" in supported
def test_max_completion_tokens_maps_to_max_tokens(self):
config = ProviderConfigManager.get_provider_chat_config(model="apodex-1.1", provider=LlmProviders.APODEX)
params = config.map_openai_params({"max_completion_tokens": 512}, {}, "apodex-1.1", False)
assert params["max_tokens"] == 512
assert "max_completion_tokens" not in params
class TestApodexAnthropicMessages:
"""Apodex serves POST /v1/messages natively, so the payload is forwarded untranslated."""
def test_native_passthrough_config(self):
config = ProviderConfigManager.get_provider_anthropic_messages_config(
model="apodex-1.1", provider=LlmProviders.APODEX
)
assert config is not None
assert type(config).__name__ == "JSONProviderAnthropicMessagesConfig"
assert (
config.get_complete_url(
api_base=None, api_key=None, model="apodex-1.1", optional_params={}, litellm_params={}
)
== "https://api.apodex.ai/v1/messages"
)
def test_headers_use_provider_api_key(self):
config = ProviderConfigManager.get_provider_anthropic_messages_config(
model="apodex-1.1", provider=LlmProviders.APODEX
)
assert config is not None
headers, _ = config.validate_anthropic_messages_environment(
headers={}, model="apodex-1.1", messages=[], optional_params={}, litellm_params={}
)
assert headers["authorization"] == "Bearer sk-apodex-test"
assert headers["anthropic-version"] == "2023-06-01"
class TestApodexModelMetadata:
@pytest.fixture(scope="class")
def model_cost(self) -> dict:
with open(REPO_ROOT / "model_prices_and_context_window.json") as f:
return json.load(f)
def test_core_model_pricing(self, model_cost: dict):
info = model_cost["apodex/apodex-1.1"]
assert info["litellm_provider"] == "apodex"
assert info["mode"] == "chat"
assert info["max_input_tokens"] == 262144
assert info["input_cost_per_token"] == 3e-07
assert info["cache_read_input_token_cost"] == 3e-08
assert info["output_cost_per_token"] == 3e-06
# Requests over 200K input tokens are billed at 2x across every tier
assert info["input_cost_per_token_above_200k_tokens"] == 6e-07
assert info["cache_read_input_token_cost_above_200k_tokens"] == 6e-08
assert info["output_cost_per_token_above_200k_tokens"] == 6e-06
assert info["supports_prompt_caching"] is True
assert info["supports_function_calling"] is True
assert info["supported_endpoints"] == ["/v1/chat/completions", "/v1/responses", "/v1/messages"]
def test_deep_research_model_pricing(self, model_cost: dict):
info = model_cost["apodex/apodex-1-1-deep-research"]
assert info["max_input_tokens"] == 131072
assert info["max_output_tokens"] == 65536
assert info["input_cost_per_token"] == 5e-06
assert info["output_cost_per_token"] == 2e-05
assert info["supports_function_calling"] is False
assert info["supports_response_schema"] is False
assert info["supports_prompt_caching"] is False
assert info["supports_web_search"] is True
assert info["supported_endpoints"] == ["/v1/chat/completions", "/v1/responses"]
def test_every_apodex_model_is_registered(self, model_cost: dict):
assert {key for key in model_cost if key.startswith("apodex/")} == {
"apodex/apodex-1.1",
"apodex/apodex-1.1-mini",
"apodex/apodex-1-1-deep-research",
"apodex/apodex-1-1-deep-solve",
"apodex/apodex-1-1-deep-discover",
"apodex/apodex-1-0-deep-research",
"apodex/apodex-1-0-deep-solve",
"apodex/apodex-1-0-deep-discover",
}
def test_backup_cost_map_in_sync(self, model_cost: dict):
with open(REPO_ROOT / "litellm" / "model_prices_and_context_window_backup.json") as f:
backup = json.load(f)
for key in (key for key in model_cost if key.startswith("apodex/")):
assert backup[key] == model_cost[key], f"{key} differs between main and backup cost maps"
def test_cost_tracks_cached_input_separately(self):
captured: dict = {}
response = litellm.completion(
model=CORE_MODEL,
messages=[{"role": "user", "content": "hi"}],
client=_openai_client(captured),
)
# 500 fresh input + 500 cached input + 100 output
expected = 500 * 3e-07 + 500 * 3e-08 + 100 * 3e-06
assert litellm.completion_cost(response, model=CORE_MODEL) == pytest.approx(expected)

View file

@ -245,63 +245,6 @@ class TestPinstripes:
assert result["temperature"] == 0.7
class TestTemperatureConstraints:
"""`constraints` in providers.json clamp temperature before the request is sent."""
@staticmethod
def _config(constraints: dict):
from litellm.llms.openai_like.dynamic_config import create_config_class
from litellm.llms.openai_like.json_loader import SimpleProviderConfig
provider = SimpleProviderConfig(
"constrained",
{
"base_url": "https://api.constrained.test/v1",
"api_key_env": "CONSTRAINED_API_KEY",
"constraints": constraints,
},
)
return create_config_class(provider)()
def test_temperature_clamped_to_max(self):
config = self._config({"temperature_max": 1.0})
result = config.map_openai_params({"temperature": 1.8}, {}, "some-model", False)
assert result["temperature"] == 1.0
def test_temperature_clamped_to_min(self):
config = self._config({"temperature_min": 0.1})
result = config.map_openai_params({"temperature": 0.0}, {}, "some-model", False)
assert result["temperature"] == 0.1
def test_temperature_within_range_is_untouched(self):
config = self._config({"temperature_min": 0.1, "temperature_max": 1.0})
result = config.map_openai_params({"temperature": 0.7}, {}, "some-model", False)
assert result["temperature"] == 0.7
def test_temperature_floor_applies_only_when_n_gt_1(self):
config = self._config({"temperature_min_with_n_gt_1": 0.3})
single = config.map_openai_params({"temperature": 0.0, "n": 1}, {}, "some-model", False)
assert single["temperature"] == 0.0
multiple = config.map_openai_params({"temperature": 0.0, "n": 2}, {}, "some-model", False)
assert multiple["temperature"] == 0.3
def test_no_constraints_leaves_temperature_alone(self):
config = self._config({})
result = config.map_openai_params({"temperature": 1.9}, {}, "some-model", False)
assert result["temperature"] == 1.9
def test_caller_optional_params_are_not_mutated(self):
config = self._config({"temperature_max": 1.0})
optional_params = {"temperature": 1.8}
result = config.map_openai_params({"max_tokens": 10}, optional_params, "some-model", False)
assert result["temperature"] == 1.0
assert result["max_tokens"] == 10
assert optional_params == {"temperature": 1.8}
class TestDarkbloom:
def test_darkbloom_json_config_exists(self):
from litellm.llms.openai_like.json_loader import JSONProviderRegistry

View file

@ -27,10 +27,10 @@
"limit": 0
},
"LIT010": {
"limit": 16707
"limit": 16715
},
"LIT011": {
"limit": 5589
"limit": 5593
},
"LIT012": {
"limit": 4519