mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
refactor(apodex): move to a Python provider with model-aware transformations
The JSON provider path applies one contract to a whole provider, which is wrong for Apodex: its two model families take different parameters. Replaces the providers.json entry with litellm/llms/apodex/, reverting the shared openai_like machinery to its original state. /v1/responses is now keyed off the model. Core models are a stateless subset, so store is pinned false and previous_response_id / background are rejected rather than passed upstream to fail with a 400. The deep research tiers keep all three, so background survives a client disconnect. /v1/messages resolves per model too. Apodex serves the protocol natively for the core models only, so the deep research tiers get no native config and fall back to translation instead of hitting a path that does not serve them. Chat completions pin stream to false for both families, drop tool params on the deep research tiers, and rename max_completion_tokens to max_tokens. The responses config also stops inheriting OpenAI's OPENAI_API_KEY fallback, which would otherwise forward an unrelated OpenAI key to Apodex. Tests live under tests/test_litellm/llms/apodex/ and touch no existing test file.
This commit is contained in:
parent
20ef10e920
commit
3b4ff9127b
20 changed files with 1001 additions and 438 deletions
|
|
@ -99,7 +99,7 @@
|
|||
"limit": 0
|
||||
},
|
||||
"reportUnknownArgumentType": {
|
||||
"limit": 44774
|
||||
"limit": 44776
|
||||
},
|
||||
"reportUnknownLambdaType": {
|
||||
"limit": 113
|
||||
|
|
@ -111,7 +111,7 @@
|
|||
"limit": 19967
|
||||
},
|
||||
"reportUnknownVariableType": {
|
||||
"limit": 30879
|
||||
"limit": 30881
|
||||
},
|
||||
"reportUnnecessaryCast": {
|
||||
"limit": 117
|
||||
|
|
|
|||
|
|
@ -638,6 +638,7 @@ snowflake_models: Set = set()
|
|||
gradient_ai_models: Set = set()
|
||||
llama_models: Set = set()
|
||||
nscale_models: Set = set()
|
||||
apodex_models: Set = set()
|
||||
nebius_models: Set = set()
|
||||
nebius_embedding_models: Set = set()
|
||||
aiml_models: Set = set()
|
||||
|
|
@ -828,6 +829,8 @@ def _populate_provider_model_sets(model_cost_map: Dict) -> None:
|
|||
llama_models.add(key)
|
||||
elif value.get("litellm_provider") == "nscale":
|
||||
nscale_models.add(key)
|
||||
elif value.get("litellm_provider") == "apodex":
|
||||
apodex_models.add(key)
|
||||
elif value.get("litellm_provider") == "azure_ai":
|
||||
azure_ai_models.add(key)
|
||||
elif value.get("litellm_provider") == "voyage":
|
||||
|
|
@ -1052,6 +1055,7 @@ model_list = list(
|
|||
| llama_models
|
||||
| featherless_ai_models
|
||||
| nscale_models
|
||||
| apodex_models
|
||||
| deepgram_models
|
||||
| elevenlabs_models
|
||||
| dashscope_models
|
||||
|
|
@ -1156,6 +1160,7 @@ def _build_models_by_provider() -> dict:
|
|||
"gradient_ai": gradient_ai_models,
|
||||
"meta_llama": llama_models,
|
||||
"nscale": nscale_models,
|
||||
"apodex": apodex_models,
|
||||
"featherless_ai": featherless_ai_models,
|
||||
"deepgram": deepgram_models,
|
||||
"elevenlabs": elevenlabs_models,
|
||||
|
|
@ -1782,6 +1787,9 @@ if TYPE_CHECKING:
|
|||
from .llms.perplexity.responses.transformation import (
|
||||
PerplexityResponsesConfig as PerplexityResponsesConfig,
|
||||
)
|
||||
from .llms.apodex.responses.transformation import (
|
||||
ApodexResponsesConfig as ApodexResponsesConfig,
|
||||
)
|
||||
from .llms.databricks.responses.transformation import (
|
||||
DatabricksResponsesAPIConfig as DatabricksResponsesAPIConfig,
|
||||
)
|
||||
|
|
@ -1855,6 +1863,7 @@ if TYPE_CHECKING:
|
|||
PerplexityChatConfig as _PerplexityChatConfig,
|
||||
)
|
||||
from .llms.nscale.chat.transformation import NscaleConfig as _NscaleConfig
|
||||
from .llms.apodex.chat.transformation import ApodexChatConfig as _ApodexChatConfig
|
||||
from .llms.watsonx.chat.transformation import (
|
||||
IBMWatsonXChatConfig as _IBMWatsonXChatConfig,
|
||||
)
|
||||
|
|
@ -1890,6 +1899,7 @@ if TYPE_CHECKING:
|
|||
AzureOpenAIO1Config: Type[_AzureOpenAIO1Config]
|
||||
PerplexityChatConfig: Type[_PerplexityChatConfig]
|
||||
NscaleConfig: Type[_NscaleConfig]
|
||||
ApodexChatConfig: Type[_ApodexChatConfig]
|
||||
IBMWatsonXChatConfig: Type[_IBMWatsonXChatConfig]
|
||||
IBMWatsonXAIConfig: Type[_IBMWatsonXAIConfig]
|
||||
LiteLLMProxyChatConfig: Type[_LiteLLMProxyChatConfig]
|
||||
|
|
|
|||
|
|
@ -238,6 +238,7 @@ LLM_CONFIG_NAMES: Final = (
|
|||
"HostedVLLMResponsesAPIConfig",
|
||||
"VolcEngineResponsesAPIConfig",
|
||||
"PerplexityResponsesConfig",
|
||||
"ApodexResponsesConfig",
|
||||
"DatabricksResponsesAPIConfig",
|
||||
"OpenRouterResponsesAPIConfig",
|
||||
"BedrockMantleResponsesAPIConfig",
|
||||
|
|
@ -291,6 +292,7 @@ LLM_CONFIG_NAMES: Final = (
|
|||
"LmStudioEmbeddingConfig",
|
||||
"NscaleConfig",
|
||||
"PerplexityChatConfig",
|
||||
"ApodexChatConfig",
|
||||
"AzureOpenAIO1Config",
|
||||
"IBMWatsonXAIConfig",
|
||||
"IBMWatsonXChatConfig",
|
||||
|
|
@ -961,6 +963,10 @@ _LLM_CONFIGS_IMPORT_MAP: Final = {
|
|||
".llms.perplexity.responses.transformation",
|
||||
"PerplexityResponsesConfig",
|
||||
),
|
||||
"ApodexResponsesConfig": (
|
||||
".llms.apodex.responses.transformation",
|
||||
"ApodexResponsesConfig",
|
||||
),
|
||||
"DatabricksResponsesAPIConfig": (
|
||||
".llms.databricks.responses.transformation",
|
||||
"DatabricksResponsesAPIConfig",
|
||||
|
|
@ -1110,6 +1116,7 @@ _LLM_CONFIGS_IMPORT_MAP: Final = {
|
|||
".llms.perplexity.chat.transformation",
|
||||
"PerplexityChatConfig",
|
||||
),
|
||||
"ApodexChatConfig": (".llms.apodex.chat.transformation", "ApodexChatConfig"),
|
||||
"AzureOpenAIO1Config": (
|
||||
".llms.azure.chat.o_series_transformation",
|
||||
"AzureOpenAIO1Config",
|
||||
|
|
|
|||
|
|
@ -824,7 +824,7 @@ openai_compatible_providers: Final[list] = [
|
|||
"pinstripes", # Pinstripes - JSON-configured provider
|
||||
"darkbloom",
|
||||
"meta", # Meta Model API (Muse Spark) - JSON-configured provider
|
||||
"apodex", # Apodex - JSON-configured provider
|
||||
"apodex",
|
||||
]
|
||||
openai_text_completion_compatible_providers: Final[list] = [ # providers that support `/v1/completions`
|
||||
"together_ai",
|
||||
|
|
|
|||
|
|
@ -349,9 +349,10 @@ def get_llm_provider(
|
|||
elif endpoint == "https://api.meta.ai/v1":
|
||||
custom_llm_provider = "meta"
|
||||
dynamic_api_key = get_secret_str("META_API_KEY")
|
||||
elif endpoint == "https://api.apodex.ai/v1":
|
||||
custom_llm_provider = "apodex"
|
||||
dynamic_api_key = get_secret_str("APODEX_API_KEY")
|
||||
elif endpoint == litellm.ApodexChatConfig.API_BASE_URL:
|
||||
custom_llm_provider = "apodex" # rebind-ok: dispatch chain resolves in place
|
||||
# rebind-ok: dispatch chain resolves in place
|
||||
dynamic_api_key = litellm.ApodexChatConfig.get_api_key()
|
||||
|
||||
if api_base is not None and not isinstance(api_base, str):
|
||||
raise Exception(f"api base needs to be a string. api_base={api_base}")
|
||||
|
|
@ -757,6 +758,9 @@ def _get_openai_compatible_provider_info(
|
|||
api_base,
|
||||
dynamic_api_key,
|
||||
) = litellm.NscaleConfig()._get_openai_compatible_provider_info(api_base=api_base, api_key=api_key)
|
||||
elif custom_llm_provider == "apodex":
|
||||
api_base = litellm.ApodexChatConfig.get_api_base(api_base) # rebind-ok: dispatch chain resolves in place
|
||||
dynamic_api_key = litellm.ApodexChatConfig.get_api_key(api_key) # rebind-ok: resolved in place
|
||||
elif custom_llm_provider == "heroku":
|
||||
(
|
||||
api_base,
|
||||
|
|
|
|||
119
litellm/llms/apodex/chat/transformation.py
Normal file
119
litellm/llms/apodex/chat/transformation.py
Normal file
|
|
@ -0,0 +1,119 @@
|
|||
"""
|
||||
Apodex chat completions — OpenAI-compatible, with two provider quirks:
|
||||
|
||||
- `stream` defaults to true upstream, so a non-streaming call has to say so
|
||||
explicitly or Apodex answers with SSE that a plain call cannot parse
|
||||
- the Deep Research tiers ignore sampling parameters and reject OpenAI-style
|
||||
tools; only the core models take them
|
||||
|
||||
Ref: https://platform.apodex.ai/docs/chat-completions
|
||||
"""
|
||||
|
||||
from collections.abc import Mapping
|
||||
from typing import Final
|
||||
|
||||
from litellm.llms.openai.chat.gpt_transformation import OpenAIGPTConfig
|
||||
|
||||
from ..common_utils import (
|
||||
APODEX_API_BASE_URL,
|
||||
get_apodex_api_base,
|
||||
get_apodex_api_key,
|
||||
is_deep_research_model,
|
||||
)
|
||||
|
||||
_DEEP_RESEARCH_PARAMS: Final = (
|
||||
"max_tokens",
|
||||
"max_completion_tokens",
|
||||
"stream",
|
||||
"stream_options",
|
||||
"extra_headers",
|
||||
"max_retries",
|
||||
)
|
||||
|
||||
_CORE_PARAMS: Final = (
|
||||
*_DEEP_RESEARCH_PARAMS,
|
||||
"temperature",
|
||||
"top_p",
|
||||
"stop",
|
||||
"seed",
|
||||
"n",
|
||||
"tools",
|
||||
"tool_choice",
|
||||
"function_call",
|
||||
"functions",
|
||||
"parallel_tool_calls",
|
||||
)
|
||||
|
||||
|
||||
class ApodexChatConfig(OpenAIGPTConfig):
|
||||
"""
|
||||
Reference: https://platform.apodex.ai/docs
|
||||
API Key: APODEX_API_KEY
|
||||
Default API Base: https://api.apodex.ai/v1
|
||||
"""
|
||||
|
||||
API_BASE_URL = APODEX_API_BASE_URL
|
||||
|
||||
@property
|
||||
def custom_llm_provider(self) -> str | None:
|
||||
return "apodex"
|
||||
|
||||
@staticmethod
|
||||
def get_api_key(api_key: str | None = None) -> str | None:
|
||||
return get_apodex_api_key(api_key)
|
||||
|
||||
@staticmethod
|
||||
def get_api_base(api_base: str | None = None) -> str | None:
|
||||
return get_apodex_api_base(api_base)
|
||||
|
||||
def _get_openai_compatible_provider_info(
|
||||
self, api_base: str | None, api_key: str | None
|
||||
) -> tuple[str | None, str | None]:
|
||||
return get_apodex_api_base(api_base), get_apodex_api_key(api_key)
|
||||
|
||||
def get_supported_openai_params(self, model: str) -> list: # mutable-ok: matches the base-class signature
|
||||
supported: Final = _DEEP_RESEARCH_PARAMS if is_deep_research_model(model) else _CORE_PARAMS
|
||||
return list(supported) # mutable-ok: matches the base-class signature
|
||||
|
||||
def map_openai_params(
|
||||
self,
|
||||
non_default_params: dict, # mutable-ok: matches the base-class signature
|
||||
optional_params: dict, # mutable-ok: matches the base-class signature
|
||||
model: str,
|
||||
drop_params: bool,
|
||||
) -> dict: # mutable-ok: matches the base-class signature
|
||||
mapped: Final = super().map_openai_params(
|
||||
non_default_params=non_default_params,
|
||||
optional_params=optional_params,
|
||||
model=model,
|
||||
drop_params=drop_params,
|
||||
)
|
||||
|
||||
# Apodex documents max_tokens only.
|
||||
renamed: Final = (
|
||||
mapped
|
||||
if "max_completion_tokens" not in mapped
|
||||
else { # mutable-ok: JSON request body
|
||||
**{ # mutable-ok: JSON request body
|
||||
key: value for key, value in mapped.items() if key != "max_completion_tokens"
|
||||
},
|
||||
"max_tokens": mapped["max_completion_tokens"],
|
||||
}
|
||||
)
|
||||
|
||||
if renamed.get("stream"):
|
||||
return renamed
|
||||
|
||||
# The OpenAI SDK drops `stream` from the body when it is false, which would
|
||||
# leave Apodex on its streaming default. extra_body is merged into the
|
||||
# request body by the SDK, so it survives that drop.
|
||||
requested_extra_body: Final = renamed.get("extra_body")
|
||||
extra_body: Final = (
|
||||
requested_extra_body
|
||||
if isinstance(requested_extra_body, Mapping)
|
||||
else {} # mutable-ok: JSON request body
|
||||
)
|
||||
return { # mutable-ok: JSON request body
|
||||
**renamed,
|
||||
"extra_body": {"stream": False, **extra_body}, # mutable-ok: JSON request body
|
||||
}
|
||||
36
litellm/llms/apodex/common_utils.py
Normal file
36
litellm/llms/apodex/common_utils.py
Normal file
|
|
@ -0,0 +1,36 @@
|
|||
"""
|
||||
Shared helpers for the Apodex provider.
|
||||
|
||||
Apodex serves two model families on one base URL, and the model id picks which
|
||||
contract applies. Core models (apodex-1.1, apodex-1.1-mini) are plain inference
|
||||
with native sampling parameters. The Deep Research tiers run an agent that
|
||||
plans, searches and iterates, so they ignore sampling parameters, reject
|
||||
OpenAI-style tools, and keep server-side state.
|
||||
|
||||
Ref: https://platform.apodex.ai/docs/models
|
||||
"""
|
||||
|
||||
from typing import Final
|
||||
|
||||
from litellm.secret_managers.main import get_secret_str
|
||||
|
||||
APODEX_API_BASE_URL: Final = "https://api.apodex.ai/v1"
|
||||
|
||||
_DEEP_RESEARCH_MARKER: Final = "-deep-"
|
||||
|
||||
|
||||
def strip_provider_prefix(model: str) -> str:
|
||||
return model.rpartition("/")[2]
|
||||
|
||||
|
||||
def is_deep_research_model(model: str) -> bool:
|
||||
"""True for the Deep Research / Solve / Discover tiers, e.g. apodex-1-1-deep-solve."""
|
||||
return _DEEP_RESEARCH_MARKER in strip_provider_prefix(model)
|
||||
|
||||
|
||||
def get_apodex_api_key(api_key: str | None = None) -> str | None:
|
||||
return api_key or get_secret_str("APODEX_API_KEY")
|
||||
|
||||
|
||||
def get_apodex_api_base(api_base: str | None = None) -> str:
|
||||
return api_base or get_secret_str("APODEX_API_BASE") or APODEX_API_BASE_URL
|
||||
52
litellm/llms/apodex/messages/transformation.py
Normal file
52
litellm/llms/apodex/messages/transformation.py
Normal file
|
|
@ -0,0 +1,52 @@
|
|||
"""
|
||||
Apodex Anthropic Messages — native passthrough for the core models only.
|
||||
|
||||
Apodex implements the Anthropic protocol itself at POST /v1/messages and serves
|
||||
the core models there, so the payload is forwarded untranslated and
|
||||
Anthropic-only features such as `thinking` and `cache_control` survive. The Deep
|
||||
Research tiers are not served on that path, so `ProviderConfigManager` hands back
|
||||
no config for them and they fall back to LiteLLM's Anthropic-to-chat-completions
|
||||
translation.
|
||||
|
||||
Ref: https://platform.apodex.ai/docs/anthropic-messages
|
||||
"""
|
||||
|
||||
from litellm.llms.openai_like.messages.transformation import (
|
||||
OpenAILikeAnthropicMessagesConfig,
|
||||
)
|
||||
|
||||
from ..common_utils import get_apodex_api_base, get_apodex_api_key
|
||||
|
||||
|
||||
class ApodexAnthropicMessagesConfig(OpenAILikeAnthropicMessagesConfig):
|
||||
@property
|
||||
def custom_llm_provider(self) -> str | None:
|
||||
return "apodex"
|
||||
|
||||
def should_strip_billing_metadata(self) -> bool:
|
||||
return True
|
||||
|
||||
def validate_anthropic_messages_environment(
|
||||
self,
|
||||
headers: dict[str, str], # mutable-ok: matches the base-class signature
|
||||
model: str,
|
||||
messages: list[object], # mutable-ok: matches the base-class signature
|
||||
optional_params: dict, # mutable-ok: matches the base-class signature
|
||||
litellm_params: dict, # mutable-ok: matches the base-class signature
|
||||
api_key: str | None = None,
|
||||
api_base: str | None = None,
|
||||
) -> tuple[dict[str, str], str | None]: # mutable-ok: matches the base-class signature
|
||||
"""Fill in the Apodex credentials and base URL.
|
||||
|
||||
The returned api_base is what the handler hands to get_complete_url, so
|
||||
resolving it here is enough to reach the native endpoint.
|
||||
"""
|
||||
return super().validate_anthropic_messages_environment(
|
||||
headers=headers,
|
||||
model=model,
|
||||
messages=messages,
|
||||
optional_params=optional_params,
|
||||
litellm_params=litellm_params,
|
||||
api_key=get_apodex_api_key(api_key),
|
||||
api_base=get_apodex_api_base(api_base),
|
||||
)
|
||||
100
litellm/llms/apodex/responses/transformation.py
Normal file
100
litellm/llms/apodex/responses/transformation.py
Normal file
|
|
@ -0,0 +1,100 @@
|
|||
"""
|
||||
Apodex Responses API — OpenAI-compatible, with a model-aware parameter contract.
|
||||
|
||||
Apodex serves /v1/responses for both model families but they accept different
|
||||
subsets, so the restrictions here are keyed off the model rather than applied
|
||||
provider-wide:
|
||||
|
||||
- core models are a stateless subset: `store` is forced to false, and
|
||||
`previous_response_id` or `background` come back as HTTP 400
|
||||
- the Deep Research tiers keep server-side state, so they take all three
|
||||
- both default `stream` to true, so a non-streaming call has to say so
|
||||
|
||||
Ref: https://platform.apodex.ai/docs/responses-api
|
||||
https://platform.apodex.ai/docs/models
|
||||
"""
|
||||
|
||||
from collections.abc import Mapping
|
||||
from typing import Final
|
||||
|
||||
import litellm
|
||||
from litellm.llms.openai.responses.transformation import OpenAIResponsesAPIConfig
|
||||
from litellm.types.llms.openai import ResponsesAPIOptionalRequestParams
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
from litellm.types.utils import LlmProviders
|
||||
|
||||
from ..common_utils import get_apodex_api_key, is_deep_research_model
|
||||
|
||||
# Rejected by the core models with HTTP 400: there is no server-side conversation
|
||||
# to resume and requests are always executed inline.
|
||||
_STATEFUL_PARAMS: Final = ("previous_response_id", "background")
|
||||
|
||||
|
||||
class ApodexResponsesConfig(OpenAIResponsesAPIConfig):
|
||||
@property
|
||||
def custom_llm_provider(self) -> LlmProviders:
|
||||
return LlmProviders.APODEX
|
||||
|
||||
def validate_environment(
|
||||
self,
|
||||
headers: dict, # mutable-ok: matches the base-class signature
|
||||
model: str,
|
||||
litellm_params: GenericLiteLLMParams | None,
|
||||
) -> dict: # mutable-ok: matches the base-class signature
|
||||
"""Resolve the Apodex key rather than inheriting OpenAI's OPENAI_API_KEY fallback,
|
||||
which would otherwise forward an unrelated OpenAI key to Apodex."""
|
||||
resolved_params: Final = litellm_params or GenericLiteLLMParams()
|
||||
api_key: Final = get_apodex_api_key(resolved_params.api_key)
|
||||
if api_key is None:
|
||||
return headers
|
||||
return { # mutable-ok: matches the base-class signature
|
||||
**headers,
|
||||
"Content-Type": "application/json",
|
||||
"Authorization": f"Bearer {api_key}",
|
||||
}
|
||||
|
||||
def get_supported_openai_params(self, model: str) -> list: # mutable-ok: matches the base-class signature
|
||||
inherited: Final = super().get_supported_openai_params(model)
|
||||
if is_deep_research_model(model):
|
||||
return inherited
|
||||
return [ # mutable-ok: matches the base-class signature
|
||||
param for param in inherited if param not in _STATEFUL_PARAMS
|
||||
]
|
||||
|
||||
def map_openai_params(
|
||||
self,
|
||||
response_api_optional_params: ResponsesAPIOptionalRequestParams,
|
||||
model: str,
|
||||
drop_params: bool,
|
||||
) -> dict: # mutable-ok: matches the base-class signature
|
||||
mapped: Final = super().map_openai_params(
|
||||
response_api_optional_params=response_api_optional_params,
|
||||
model=model,
|
||||
drop_params=drop_params,
|
||||
)
|
||||
|
||||
stateless: Final = (
|
||||
mapped
|
||||
if is_deep_research_model(model)
|
||||
else self._enforce_stateless(mapped, model=model, drop_params=drop_params)
|
||||
)
|
||||
|
||||
if stateless.get("stream"):
|
||||
return {**stateless} # mutable-ok: JSON request body
|
||||
return {**stateless, "stream": False} # mutable-ok: JSON request body
|
||||
|
||||
@staticmethod
|
||||
def _enforce_stateless(params: Mapping[str, object], model: str, drop_params: bool) -> Mapping[str, object]:
|
||||
"""Core models only: drop what the stateless subset rejects and pin store to false."""
|
||||
if params.get("store") is True and not (drop_params or litellm.drop_params):
|
||||
raise litellm.UnsupportedParamsError(
|
||||
message=(
|
||||
f"apodex model {model} does not support store=True on /v1/responses: the endpoint is a "
|
||||
"stateless subset. To drop this, set `litellm.drop_params = True`"
|
||||
),
|
||||
status_code=400,
|
||||
)
|
||||
kept: Final = { # mutable-ok: JSON request body
|
||||
key: value for key, value in params.items() if key not in _STATEFUL_PARAMS
|
||||
}
|
||||
return {**kept, "store": False} # mutable-ok: JSON request body
|
||||
|
|
@ -59,15 +59,7 @@ That's it! The provider will be automatically loaded and available.
|
|||
|
||||
// Optional: Special handling flags
|
||||
"special_handling": {
|
||||
"convert_content_list_to_string": true,
|
||||
|
||||
// Send "stream": false explicitly instead of omitting it. Needed by
|
||||
// providers whose /v1/chat/completions and /v1/responses default to
|
||||
// streaming, where omitting the field returns SSE to a non-streaming call
|
||||
"send_explicit_stream_false": true,
|
||||
|
||||
// Always send "store": false on /v1/responses
|
||||
"force_store_false": true
|
||||
"convert_content_list_to_string": true
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -2,7 +2,7 @@
|
|||
Dynamic configuration class generator for JSON-based providers.
|
||||
"""
|
||||
|
||||
from collections.abc import Coroutine, Mapping
|
||||
from collections.abc import Coroutine
|
||||
from typing import Any, Final, Literal, overload
|
||||
|
||||
from litellm._logging import verbose_logger
|
||||
|
|
@ -17,17 +17,6 @@ from litellm.types.llms.openai import AllMessageValues
|
|||
from .json_loader import SimpleProviderConfig
|
||||
|
||||
|
||||
def _clamp_temperature(temperature: float, n: int, constraints: Mapping[str, float]) -> float:
|
||||
capped: Final = (
|
||||
min(temperature, constraints["temperature_max"]) if "temperature_max" in constraints else temperature
|
||||
)
|
||||
floored: Final = max(capped, constraints["temperature_min"]) if "temperature_min" in constraints else capped
|
||||
floor_for_multiple_choices: Final = constraints.get("temperature_min_with_n_gt_1")
|
||||
if n > 1 and floor_for_multiple_choices is not None:
|
||||
return max(floored, floor_for_multiple_choices)
|
||||
return floored
|
||||
|
||||
|
||||
def create_config_class(provider: SimpleProviderConfig):
|
||||
"""Generate config class dynamically from JSON configuration"""
|
||||
|
||||
|
|
@ -142,36 +131,37 @@ def create_config_class(provider: SimpleProviderConfig):
|
|||
"""Apply parameter mappings and constraints"""
|
||||
|
||||
supported_params: Final = self.get_supported_openai_params(model)
|
||||
mapped: Final = {
|
||||
**optional_params,
|
||||
**{
|
||||
provider.param_mappings.get(param, param): value
|
||||
for param, value in non_default_params.items()
|
||||
if param in provider.param_mappings or param in supported_params
|
||||
},
|
||||
}
|
||||
|
||||
constrained: Final = (
|
||||
mapped
|
||||
if "temperature" not in mapped
|
||||
else {
|
||||
**mapped,
|
||||
"temperature": _clamp_temperature(
|
||||
temperature=mapped["temperature"],
|
||||
n=mapped.get("n", 1),
|
||||
constraints=provider.constraints,
|
||||
),
|
||||
}
|
||||
)
|
||||
# Apply supported params
|
||||
for param, value in non_default_params.items():
|
||||
# Check parameter mappings first
|
||||
if param in provider.param_mappings:
|
||||
optional_params[provider.param_mappings[param]] = value
|
||||
elif param in supported_params:
|
||||
optional_params[param] = value
|
||||
|
||||
# The OpenAI SDK omits `stream` entirely when it is false, which makes
|
||||
# stream-by-default providers answer a non-streaming call with SSE. Pin it
|
||||
# on the wire through extra_body, which the SDK merges into the request body.
|
||||
if not provider.special_handling.get("send_explicit_stream_false") or constrained.get("stream"):
|
||||
return constrained
|
||||
requested_extra_body: Final = constrained.get("extra_body")
|
||||
extra_body: Final[dict] = requested_extra_body if isinstance(requested_extra_body, dict) else {}
|
||||
return {**constrained, "extra_body": {"stream": False, **extra_body}}
|
||||
# Apply temperature constraints if present
|
||||
if "temperature" in optional_params:
|
||||
temp = optional_params["temperature"]
|
||||
constraints: Final = provider.constraints
|
||||
|
||||
# Clamp to max
|
||||
if "temperature_max" in constraints:
|
||||
temp = min(temp, constraints["temperature_max"])
|
||||
|
||||
# Clamp to min
|
||||
if "temperature_min" in constraints:
|
||||
temp = max(temp, constraints["temperature_min"])
|
||||
|
||||
# Special case: temperature_min_with_n_gt_1
|
||||
if "temperature_min_with_n_gt_1" in constraints:
|
||||
n: Final = optional_params.get("n", 1)
|
||||
if n > 1 and temp < constraints["temperature_min_with_n_gt_1"]:
|
||||
temp = constraints["temperature_min_with_n_gt_1"]
|
||||
|
||||
optional_params["temperature"] = temp
|
||||
|
||||
return optional_params
|
||||
|
||||
@property
|
||||
def custom_llm_provider(self) -> str | None:
|
||||
|
|
@ -242,8 +232,6 @@ def create_responses_config_class(provider: SimpleProviderConfig):
|
|||
) -> dict:
|
||||
if provider.special_handling.get("force_store_false"):
|
||||
response_api_optional_request_params["store"] = False
|
||||
if provider.special_handling.get("send_explicit_stream_false"):
|
||||
response_api_optional_request_params.setdefault("stream", False)
|
||||
return super().transform_responses_api_request(
|
||||
model=model,
|
||||
input=input,
|
||||
|
|
|
|||
|
|
@ -175,18 +175,6 @@
|
|||
"base_class": "openai_gpt",
|
||||
"supported_endpoints": ["/v1/chat/completions", "/v1/responses", "/v1/messages"]
|
||||
},
|
||||
"apodex": {
|
||||
"base_url": "https://api.apodex.ai/v1",
|
||||
"api_key_env": "APODEX_API_KEY",
|
||||
"api_base_env": "APODEX_API_BASE",
|
||||
"param_mappings": {
|
||||
"max_completion_tokens": "max_tokens"
|
||||
},
|
||||
"supported_endpoints": ["/v1/chat/completions", "/v1/responses", "/v1/messages"],
|
||||
"special_handling": {
|
||||
"send_explicit_stream_false": true
|
||||
}
|
||||
},
|
||||
"pinstripes": {
|
||||
"base_url": "https://pinstripes.io/v1",
|
||||
"api_key_env": "PINSTRIPES_API_KEY",
|
||||
|
|
|
|||
|
|
@ -7928,6 +7928,7 @@ class ProviderConfigManager:
|
|||
),
|
||||
LlmProviders.GRADIENT_AI: (lambda: litellm.GradientAIConfig(), False),
|
||||
LlmProviders.NSCALE: (lambda: litellm.NscaleConfig(), False),
|
||||
LlmProviders.APODEX: (lambda: litellm.ApodexChatConfig(), False),
|
||||
LlmProviders.HEROKU: (lambda: litellm.HerokuChatConfig(), False),
|
||||
LlmProviders.OCI: (lambda: litellm.OCIChatConfig(), False),
|
||||
LlmProviders.HYPERBOLIC: (lambda: litellm.HyperbolicChatConfig(), False),
|
||||
|
|
@ -8255,6 +8256,17 @@ class ProviderConfigManager:
|
|||
)
|
||||
|
||||
return DeepSeekAnthropicMessagesConfig()
|
||||
elif litellm.LlmProviders.APODEX == provider:
|
||||
from litellm.llms.apodex.common_utils import is_deep_research_model
|
||||
from litellm.llms.apodex.messages.transformation import (
|
||||
ApodexAnthropicMessagesConfig,
|
||||
)
|
||||
|
||||
# Apodex only serves the core models on its native /v1/messages path; the
|
||||
# deep research tiers get no config so they fall back to translation.
|
||||
if is_deep_research_model(model):
|
||||
return None
|
||||
return ApodexAnthropicMessagesConfig()
|
||||
elif litellm.LlmProviders.TENCENT == provider:
|
||||
from litellm.llms.tencent.messages.transformation import (
|
||||
TencentAnthropicMessagesConfig,
|
||||
|
|
@ -8440,6 +8452,8 @@ class ProviderConfigManager:
|
|||
return litellm.ManusResponsesAPIConfig()
|
||||
elif litellm.LlmProviders.PERPLEXITY == provider:
|
||||
return litellm.PerplexityResponsesConfig()
|
||||
elif litellm.LlmProviders.APODEX == provider:
|
||||
return litellm.ApodexResponsesConfig()
|
||||
elif litellm.LlmProviders.DATABRICKS == provider:
|
||||
# Databricks Responses API is only compatible with OpenAI GPT models
|
||||
if model and "gpt" in model.lower():
|
||||
|
|
|
|||
|
|
@ -0,0 +1,209 @@
|
|||
"""
|
||||
Apodex chat completions transformation.
|
||||
"""
|
||||
|
||||
import json
|
||||
|
||||
import httpx
|
||||
import openai
|
||||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.types.utils import LlmProviders
|
||||
from litellm.utils import ProviderConfigManager
|
||||
|
||||
CORE_MODEL = "apodex/apodex-1.1"
|
||||
DEEP_RESEARCH_MODEL = "apodex/apodex-1-1-deep-research"
|
||||
|
||||
CHAT_RESPONSE = {
|
||||
"id": "chatcmpl-abc123",
|
||||
"object": "chat.completion",
|
||||
"created": 1712345678,
|
||||
"model": "apodex-1.1",
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"message": {"role": "assistant", "content": "ok", "reasoning_content": "let me think"},
|
||||
"finish_reason": "stop",
|
||||
}
|
||||
],
|
||||
"usage": {
|
||||
"prompt_tokens": 1000,
|
||||
"completion_tokens": 100,
|
||||
"total_tokens": 1100,
|
||||
"prompt_tokens_details": {"cached_tokens": 500},
|
||||
},
|
||||
}
|
||||
|
||||
STREAM_BODY = (
|
||||
b'data: {"id":"chatcmpl-abc123","object":"chat.completion.chunk","created":1,"model":"apodex-1.1",'
|
||||
b'"choices":[{"index":0,"delta":{"content":"ok"},"finish_reason":null}]}\n\n'
|
||||
b"data: [DONE]\n\n"
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _apodex_env(monkeypatch: pytest.MonkeyPatch):
|
||||
"""Resolve models against the in-repo cost map, not the published one."""
|
||||
monkeypatch.setenv("APODEX_API_KEY", "sk-apodex-test")
|
||||
monkeypatch.delenv("APODEX_API_BASE", raising=False)
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
|
||||
yield
|
||||
|
||||
|
||||
def _client(captured: dict, *, stream: bool = False) -> openai.OpenAI:
|
||||
def handler(request: httpx.Request) -> httpx.Response:
|
||||
captured["url"] = str(request.url)
|
||||
captured["body"] = json.loads(request.content)
|
||||
if stream:
|
||||
return httpx.Response(200, headers={"content-type": "text/event-stream"}, content=STREAM_BODY)
|
||||
return httpx.Response(200, json=CHAT_RESPONSE)
|
||||
|
||||
return openai.OpenAI(
|
||||
api_key="sk-apodex-test",
|
||||
base_url="https://api.apodex.ai/v1",
|
||||
http_client=httpx.Client(transport=httpx.MockTransport(handler)),
|
||||
)
|
||||
|
||||
|
||||
def _chat_config(model: str):
|
||||
return ProviderConfigManager.get_provider_chat_config(model=model, provider=LlmProviders.APODEX)
|
||||
|
||||
|
||||
class TestProviderResolution:
|
||||
def test_prefixed_model_resolves_to_the_default_base(self):
|
||||
model, provider, api_key, api_base = litellm.get_llm_provider(model=CORE_MODEL)
|
||||
assert (model, provider, api_key, api_base) == (
|
||||
"apodex-1.1",
|
||||
"apodex",
|
||||
"sk-apodex-test",
|
||||
"https://api.apodex.ai/v1",
|
||||
)
|
||||
|
||||
def test_api_base_autodetection(self):
|
||||
_, provider, api_key, _ = litellm.get_llm_provider(model="apodex-1.1", api_base="https://api.apodex.ai/v1")
|
||||
assert provider == "apodex"
|
||||
assert api_key == "sk-apodex-test"
|
||||
|
||||
def test_explicit_api_base_and_key_win(self):
|
||||
_, provider, api_key, api_base = litellm.get_llm_provider(
|
||||
model=CORE_MODEL, api_base="https://gateway.internal/v1", api_key="sk-override"
|
||||
)
|
||||
assert (provider, api_key, api_base) == ("apodex", "sk-override", "https://gateway.internal/v1")
|
||||
|
||||
def test_api_base_env_override(self, monkeypatch: pytest.MonkeyPatch):
|
||||
monkeypatch.setenv("APODEX_API_BASE", "https://env.apodex.test/v1")
|
||||
_, _, _, api_base = litellm.get_llm_provider(model=CORE_MODEL)
|
||||
assert api_base == "https://env.apodex.test/v1"
|
||||
|
||||
|
||||
class TestStreamDefault:
|
||||
"""Apodex defaults `stream` to true, so a non-streaming call has to pin it to false.
|
||||
|
||||
Regression guard: the OpenAI SDK drops `stream` from the body when it is false,
|
||||
which would leave Apodex streaming SSE at a call that cannot parse it.
|
||||
"""
|
||||
|
||||
def test_non_streaming_call_pins_stream_false(self):
|
||||
captured: dict = {}
|
||||
response = litellm.completion(
|
||||
model=CORE_MODEL,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
client=_client(captured),
|
||||
)
|
||||
|
||||
assert captured["url"] == "https://api.apodex.ai/v1/chat/completions"
|
||||
assert captured["body"]["stream"] is False
|
||||
assert captured["body"]["model"] == "apodex-1.1"
|
||||
assert response.choices[0].message.reasoning_content == "let me think"
|
||||
|
||||
def test_streaming_call_sends_stream_true(self):
|
||||
captured: dict = {}
|
||||
chunks = list(
|
||||
litellm.completion(
|
||||
model=CORE_MODEL,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
stream=True,
|
||||
client=_client(captured, stream=True),
|
||||
)
|
||||
)
|
||||
|
||||
assert captured["body"]["stream"] is True
|
||||
assert chunks
|
||||
|
||||
def test_deep_research_models_pin_stream_too(self):
|
||||
captured: dict = {}
|
||||
litellm.completion(
|
||||
model=DEEP_RESEARCH_MODEL,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
client=_client(captured),
|
||||
)
|
||||
assert captured["body"]["stream"] is False
|
||||
|
||||
def test_user_supplied_extra_body_is_preserved(self):
|
||||
"""Deep research tiers reach external tools through `mcp_servers` in extra_body."""
|
||||
captured: dict = {}
|
||||
mcp_servers = [{"name": "docs", "url": "https://example.com/mcp"}]
|
||||
litellm.completion(
|
||||
model=DEEP_RESEARCH_MODEL,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
extra_body={"mcp_servers": mcp_servers},
|
||||
client=_client(captured),
|
||||
)
|
||||
|
||||
assert captured["body"]["stream"] is False
|
||||
assert captured["body"]["mcp_servers"] == mcp_servers
|
||||
|
||||
|
||||
class TestSupportedParams:
|
||||
def test_core_models_support_tools(self):
|
||||
supported = _chat_config("apodex-1.1").get_supported_openai_params("apodex-1.1")
|
||||
assert "tools" in supported
|
||||
assert "tool_choice" in supported
|
||||
assert "temperature" in supported
|
||||
assert "top_p" in supported
|
||||
|
||||
def test_deep_research_rejects_tools_and_sampling_params(self):
|
||||
"""The tiers document tools as unsupported and sampling params as ignored."""
|
||||
supported = _chat_config(DEEP_RESEARCH_MODEL).get_supported_openai_params("apodex-1-1-deep-research")
|
||||
for param in ("tools", "tool_choice", "function_call", "functions", "parallel_tool_calls"):
|
||||
assert param not in supported
|
||||
assert "temperature" not in supported
|
||||
assert "top_p" not in supported
|
||||
assert "max_tokens" in supported
|
||||
|
||||
def test_tools_on_a_deep_research_model_raise(self):
|
||||
with pytest.raises(litellm.UnsupportedParamsError, match="tools"):
|
||||
litellm.completion(
|
||||
model=DEEP_RESEARCH_MODEL,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
tools=[{"type": "function", "function": {"name": "f", "parameters": {}}}],
|
||||
client=_client({}),
|
||||
)
|
||||
|
||||
def test_max_completion_tokens_is_renamed_to_max_tokens(self):
|
||||
captured: dict = {}
|
||||
litellm.completion(
|
||||
model=CORE_MODEL,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
max_completion_tokens=512,
|
||||
client=_client(captured),
|
||||
)
|
||||
|
||||
assert captured["body"]["max_tokens"] == 512
|
||||
assert "max_completion_tokens" not in captured["body"]
|
||||
|
||||
|
||||
class TestCostTracking:
|
||||
def test_cached_input_is_billed_at_the_lower_rate(self):
|
||||
captured: dict = {}
|
||||
response = litellm.completion(
|
||||
model=CORE_MODEL,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
client=_client(captured),
|
||||
)
|
||||
|
||||
# 500 fresh input + 500 cached input + 100 output
|
||||
expected = 500 * 3e-07 + 500 * 3e-08 + 100 * 3e-06
|
||||
assert litellm.completion_cost(response, model=CORE_MODEL) == pytest.approx(expected)
|
||||
136
tests/test_litellm/llms/apodex/test_apodex_common_utils.py
Normal file
136
tests/test_litellm/llms/apodex/test_apodex_common_utils.py
Normal file
|
|
@ -0,0 +1,136 @@
|
|||
"""
|
||||
Apodex provider registration and model-family classification.
|
||||
"""
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.llms.apodex.common_utils import (
|
||||
APODEX_API_BASE_URL,
|
||||
get_apodex_api_base,
|
||||
get_apodex_api_key,
|
||||
is_deep_research_model,
|
||||
)
|
||||
from litellm.types.utils import LlmProviders
|
||||
|
||||
REPO_ROOT = Path(__file__).parents[4]
|
||||
|
||||
CORE_MODELS = ("apodex-1.1", "apodex-1.1-mini")
|
||||
DEEP_RESEARCH_MODELS = (
|
||||
"apodex-1-1-deep-research",
|
||||
"apodex-1-1-deep-solve",
|
||||
"apodex-1-1-deep-discover",
|
||||
"apodex-1-0-deep-research",
|
||||
"apodex-1-0-deep-solve",
|
||||
"apodex-1-0-deep-discover",
|
||||
)
|
||||
|
||||
|
||||
class TestModelFamily:
|
||||
"""The model id, not the provider, selects which Apodex contract applies."""
|
||||
|
||||
@pytest.mark.parametrize("model", CORE_MODELS)
|
||||
def test_core_models_are_not_deep_research(self, model: str):
|
||||
assert is_deep_research_model(model) is False
|
||||
assert is_deep_research_model(f"apodex/{model}") is False
|
||||
|
||||
@pytest.mark.parametrize("model", DEEP_RESEARCH_MODELS)
|
||||
def test_deep_research_models_are_detected(self, model: str):
|
||||
assert is_deep_research_model(model) is True
|
||||
assert is_deep_research_model(f"apodex/{model}") is True
|
||||
|
||||
def test_prefix_does_not_leak_into_classification(self):
|
||||
"""A provider prefix containing the marker must not flip a core model."""
|
||||
assert is_deep_research_model("some-deep-gateway/apodex-1.1") is False
|
||||
|
||||
|
||||
class TestCredentialResolution:
|
||||
def test_defaults(self, monkeypatch: pytest.MonkeyPatch):
|
||||
monkeypatch.delenv("APODEX_API_BASE", raising=False)
|
||||
monkeypatch.setenv("APODEX_API_KEY", "sk-env")
|
||||
|
||||
assert get_apodex_api_base(None) == APODEX_API_BASE_URL
|
||||
assert get_apodex_api_key(None) == "sk-env"
|
||||
|
||||
def test_explicit_values_win_over_env(self, monkeypatch: pytest.MonkeyPatch):
|
||||
monkeypatch.setenv("APODEX_API_BASE", "https://env.apodex.test/v1")
|
||||
monkeypatch.setenv("APODEX_API_KEY", "sk-env")
|
||||
|
||||
assert get_apodex_api_base("https://explicit.apodex.test/v1") == "https://explicit.apodex.test/v1"
|
||||
assert get_apodex_api_key("sk-explicit") == "sk-explicit"
|
||||
|
||||
def test_env_base_overrides_default(self, monkeypatch: pytest.MonkeyPatch):
|
||||
monkeypatch.setenv("APODEX_API_BASE", "https://env.apodex.test/v1")
|
||||
assert get_apodex_api_base(None) == "https://env.apodex.test/v1"
|
||||
|
||||
|
||||
class TestRegistration:
|
||||
def test_provider_enum_and_lists(self):
|
||||
assert LlmProviders.APODEX.value == "apodex"
|
||||
assert "apodex" in litellm.provider_list
|
||||
assert "apodex" in litellm.constants.openai_compatible_providers
|
||||
assert APODEX_API_BASE_URL in litellm.constants.openai_compatible_endpoints
|
||||
|
||||
def test_not_registered_as_a_json_provider(self):
|
||||
"""Apodex needs model-aware transformations, so it must not fall into the
|
||||
generic JSON path, which would shadow the Python configs in provider resolution."""
|
||||
from litellm.llms.openai_like.json_loader import JSONProviderRegistry
|
||||
|
||||
assert JSONProviderRegistry.exists("apodex") is False
|
||||
|
||||
def test_config_classes_resolve_from_the_lazy_registry(self):
|
||||
assert litellm.ApodexChatConfig().custom_llm_provider == "apodex"
|
||||
assert litellm.ApodexResponsesConfig().custom_llm_provider == LlmProviders.APODEX
|
||||
|
||||
|
||||
class TestModelMetadata:
|
||||
@pytest.fixture(scope="class")
|
||||
def model_cost(self) -> dict:
|
||||
with open(REPO_ROOT / "model_prices_and_context_window.json") as f:
|
||||
return json.load(f)
|
||||
|
||||
def test_every_apodex_model_is_registered(self, model_cost: dict):
|
||||
assert {key for key in model_cost if key.startswith("apodex/")} == {
|
||||
f"apodex/{model}" for model in (*CORE_MODELS, *DEEP_RESEARCH_MODELS)
|
||||
}
|
||||
|
||||
def test_core_model_pricing(self, model_cost: dict):
|
||||
info = model_cost["apodex/apodex-1.1"]
|
||||
assert info["litellm_provider"] == "apodex"
|
||||
assert info["mode"] == "chat"
|
||||
assert info["max_input_tokens"] == 262144
|
||||
assert info["input_cost_per_token"] == 3e-07
|
||||
assert info["cache_read_input_token_cost"] == 3e-08
|
||||
assert info["output_cost_per_token"] == 3e-06
|
||||
# Requests over 200K input tokens are billed at 2x across every tier
|
||||
assert info["input_cost_per_token_above_200k_tokens"] == 6e-07
|
||||
assert info["cache_read_input_token_cost_above_200k_tokens"] == 6e-08
|
||||
assert info["output_cost_per_token_above_200k_tokens"] == 6e-06
|
||||
assert info["supports_prompt_caching"] is True
|
||||
assert info["supports_function_calling"] is True
|
||||
assert info["supported_endpoints"] == ["/v1/chat/completions", "/v1/responses", "/v1/messages"]
|
||||
|
||||
def test_deep_research_model_pricing(self, model_cost: dict):
|
||||
info = model_cost["apodex/apodex-1-1-deep-research"]
|
||||
assert info["max_input_tokens"] == 131072
|
||||
assert info["max_output_tokens"] == 65536
|
||||
assert info["input_cost_per_token"] == 5e-06
|
||||
assert info["output_cost_per_token"] == 2e-05
|
||||
assert info["supports_function_calling"] is False
|
||||
assert info["supports_response_schema"] is False
|
||||
assert info["supports_prompt_caching"] is False
|
||||
assert info["supports_web_search"] is True
|
||||
|
||||
@pytest.mark.parametrize("model", DEEP_RESEARCH_MODELS)
|
||||
def test_deep_research_models_are_not_on_the_native_messages_path(self, model_cost: dict, model: str):
|
||||
"""Apodex serves /v1/messages for the core models only."""
|
||||
assert "/v1/messages" not in model_cost[f"apodex/{model}"]["supported_endpoints"]
|
||||
|
||||
def test_backup_cost_map_in_sync(self, model_cost: dict):
|
||||
with open(REPO_ROOT / "litellm" / "model_prices_and_context_window_backup.json") as f:
|
||||
backup = json.load(f)
|
||||
for key in (key for key in model_cost if key.startswith("apodex/")):
|
||||
assert backup[key] == model_cost[key], f"{key} differs between main and backup cost maps"
|
||||
|
|
@ -0,0 +1,98 @@
|
|||
"""
|
||||
Apodex Anthropic Messages transformation.
|
||||
|
||||
Apodex implements the Anthropic protocol natively at POST /v1/messages, but only
|
||||
serves the core models there. The Deep Research tiers must keep working on the
|
||||
same route through LiteLLM's translation instead of being handed to a path that
|
||||
would reject them.
|
||||
"""
|
||||
|
||||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.types.utils import LlmProviders
|
||||
from litellm.utils import ProviderConfigManager
|
||||
|
||||
CORE_MODELS = ("apodex-1.1", "apodex-1.1-mini")
|
||||
DEEP_RESEARCH_MODELS = (
|
||||
"apodex-1-1-deep-research",
|
||||
"apodex-1-1-deep-solve",
|
||||
"apodex-1-1-deep-discover",
|
||||
"apodex-1-0-deep-research",
|
||||
"apodex-1-0-deep-solve",
|
||||
"apodex-1-0-deep-discover",
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _apodex_env(monkeypatch: pytest.MonkeyPatch):
|
||||
monkeypatch.setenv("APODEX_API_KEY", "sk-apodex-test")
|
||||
monkeypatch.delenv("APODEX_API_BASE", raising=False)
|
||||
yield
|
||||
|
||||
|
||||
def _messages_config(model: str):
|
||||
return ProviderConfigManager.get_provider_anthropic_messages_config(model=model, provider=LlmProviders.APODEX)
|
||||
|
||||
|
||||
def _complete_url() -> str:
|
||||
"""Resolve the endpoint the way the handler does: validate first, then build the URL.
|
||||
|
||||
validate_anthropic_messages_environment returns the api_base the handler feeds
|
||||
into get_complete_url, so the two steps have to run in that order.
|
||||
"""
|
||||
config = _messages_config("apodex-1.1")
|
||||
assert config is not None
|
||||
_, api_base = config.validate_anthropic_messages_environment(
|
||||
headers={}, model="apodex-1.1", messages=[], optional_params={}, litellm_params={}
|
||||
)
|
||||
return config.get_complete_url(
|
||||
api_base=api_base, api_key=None, model="apodex-1.1", optional_params={}, litellm_params={}
|
||||
)
|
||||
|
||||
|
||||
class TestNativePassthroughRouting:
|
||||
@pytest.mark.parametrize("model", CORE_MODELS)
|
||||
def test_core_models_get_the_native_config(self, model: str):
|
||||
config = _messages_config(model)
|
||||
assert config is not None
|
||||
assert type(config).__name__ == "ApodexAnthropicMessagesConfig"
|
||||
assert config.custom_llm_provider == "apodex"
|
||||
|
||||
@pytest.mark.parametrize("model", DEEP_RESEARCH_MODELS)
|
||||
def test_deep_research_models_fall_back_to_translation(self, model: str):
|
||||
"""No native config means LiteLLM translates to chat completions, which works,
|
||||
instead of forwarding to a path Apodex does not serve for these tiers."""
|
||||
assert _messages_config(model) is None
|
||||
|
||||
|
||||
class TestNativePassthroughRequest:
|
||||
def test_url_targets_the_native_messages_path(self):
|
||||
assert _complete_url() == "https://api.apodex.ai/v1/messages"
|
||||
|
||||
def test_url_honours_an_api_base_override(self, monkeypatch: pytest.MonkeyPatch):
|
||||
monkeypatch.setenv("APODEX_API_BASE", "https://env.apodex.test/v1")
|
||||
assert _complete_url() == "https://env.apodex.test/v1/messages"
|
||||
|
||||
def test_headers_use_the_provider_api_key(self):
|
||||
config = _messages_config("apodex-1.1")
|
||||
assert config is not None
|
||||
headers, _ = config.validate_anthropic_messages_environment(
|
||||
headers={}, model="apodex-1.1", messages=[], optional_params={}, litellm_params={}
|
||||
)
|
||||
assert headers["authorization"] == "Bearer sk-apodex-test"
|
||||
assert headers["anthropic-version"] == "2023-06-01"
|
||||
assert headers["content-type"] == "application/json"
|
||||
|
||||
def test_caller_supplied_auth_header_is_not_overwritten(self):
|
||||
config = _messages_config("apodex-1.1")
|
||||
assert config is not None
|
||||
headers, _ = config.validate_anthropic_messages_environment(
|
||||
headers={"x-api-key": "sk-caller"},
|
||||
model="apodex-1.1",
|
||||
messages=[],
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
)
|
||||
assert headers["x-api-key"] == "sk-caller"
|
||||
assert "authorization" not in headers
|
||||
|
|
@ -0,0 +1,177 @@
|
|||
"""
|
||||
Apodex Responses API transformation.
|
||||
|
||||
The core models expose a stateless subset of /v1/responses while the Deep
|
||||
Research tiers keep server-side state, so the parameter contract is keyed off
|
||||
the model rather than applied provider-wide.
|
||||
"""
|
||||
|
||||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler
|
||||
from litellm.types.utils import LlmProviders
|
||||
from litellm.utils import ProviderConfigManager
|
||||
|
||||
CORE_MODEL = "apodex/apodex-1.1"
|
||||
CORE_MINI_MODEL = "apodex/apodex-1.1-mini"
|
||||
DEEP_RESEARCH_MODEL = "apodex/apodex-1-1-deep-research"
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _apodex_env(monkeypatch: pytest.MonkeyPatch):
|
||||
monkeypatch.setenv("APODEX_API_KEY", "sk-apodex-test")
|
||||
monkeypatch.delenv("APODEX_API_BASE", raising=False)
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
|
||||
monkeypatch.setattr(litellm, "drop_params", False)
|
||||
yield
|
||||
|
||||
|
||||
_SENTINEL = "apodex-request-captured"
|
||||
|
||||
|
||||
def _capture(**kwargs) -> dict:
|
||||
"""Run litellm.responses() and return the request it would have sent.
|
||||
|
||||
Validation errors raised before the request is built propagate to the caller.
|
||||
"""
|
||||
captured: dict = {}
|
||||
|
||||
class CapturingHandler(HTTPHandler):
|
||||
def post(self, *args, **post_kwargs):
|
||||
captured.update(url=post_kwargs.get("url"), body=post_kwargs.get("json"))
|
||||
raise RuntimeError(_SENTINEL)
|
||||
|
||||
try:
|
||||
litellm.responses(client=CapturingHandler(), **kwargs)
|
||||
except Exception as exc:
|
||||
if _SENTINEL not in str(exc):
|
||||
raise
|
||||
assert captured, "no request was sent"
|
||||
return captured
|
||||
|
||||
|
||||
def _responses_config(model: str):
|
||||
return ProviderConfigManager.get_provider_responses_api_config(model=model, provider=LlmProviders.APODEX)
|
||||
|
||||
|
||||
class TestConfigSelection:
|
||||
def test_python_config_is_used_for_every_apodex_model(self):
|
||||
for model in ("apodex-1.1", "apodex-1.1-mini", "apodex-1-1-deep-research"):
|
||||
config = _responses_config(model)
|
||||
assert type(config).__name__ == "ApodexResponsesConfig"
|
||||
|
||||
def test_auth_uses_the_apodex_key(self):
|
||||
config = _responses_config("apodex-1.1")
|
||||
assert config.validate_environment(headers={}, model="apodex-1.1", litellm_params=None) == {
|
||||
"Content-Type": "application/json",
|
||||
"Authorization": "Bearer sk-apodex-test",
|
||||
}
|
||||
|
||||
def test_auth_does_not_fall_back_to_an_openai_key(self, monkeypatch: pytest.MonkeyPatch):
|
||||
"""The inherited OpenAI config would forward OPENAI_API_KEY to Apodex."""
|
||||
monkeypatch.delenv("APODEX_API_KEY", raising=False)
|
||||
monkeypatch.setenv("OPENAI_API_KEY", "sk-openai-must-not-leak")
|
||||
|
||||
config = _responses_config("apodex-1.1")
|
||||
assert config.validate_environment(headers={}, model="apodex-1.1", litellm_params=None) == {}
|
||||
|
||||
def test_request_targets_the_apodex_responses_url(self):
|
||||
assert _capture(model=CORE_MODEL, input="hi")["url"] == "https://api.apodex.ai/v1/responses"
|
||||
|
||||
def test_request_honours_an_api_base_override(self, monkeypatch: pytest.MonkeyPatch):
|
||||
monkeypatch.setenv("APODEX_API_BASE", "https://env.apodex.test/v1")
|
||||
assert _capture(model=CORE_MODEL, input="hi")["url"] == "https://env.apodex.test/v1/responses"
|
||||
|
||||
|
||||
class TestStreamDefault:
|
||||
def test_non_streaming_pins_stream_false(self):
|
||||
captured = _capture(model=DEEP_RESEARCH_MODEL, input="hi")
|
||||
assert captured["url"] == "https://api.apodex.ai/v1/responses"
|
||||
assert captured["body"]["stream"] is False
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_streaming_sends_stream_true(self):
|
||||
captured: dict = {}
|
||||
|
||||
class CapturingHandler(AsyncHTTPHandler):
|
||||
async def post(self, *args, **kwargs):
|
||||
captured.update(body=kwargs.get("json"))
|
||||
raise RuntimeError("captured")
|
||||
|
||||
with pytest.raises(Exception, match="captured"):
|
||||
await litellm.aresponses(
|
||||
model=DEEP_RESEARCH_MODEL, input="hi", stream=True, client=CapturingHandler()
|
||||
)
|
||||
|
||||
assert captured["body"]["stream"] is True
|
||||
|
||||
|
||||
class TestCoreModelStatelessSubset:
|
||||
"""Apodex core models reject anything that would persist state on their side."""
|
||||
|
||||
@pytest.mark.parametrize("model", [CORE_MODEL, CORE_MINI_MODEL])
|
||||
def test_store_is_pinned_false(self, model: str):
|
||||
captured = _capture(model=model, input="hi")
|
||||
assert captured["body"]["store"] is False
|
||||
|
||||
def test_store_true_raises(self):
|
||||
with pytest.raises(litellm.UnsupportedParamsError, match="store=True"):
|
||||
_capture(model=CORE_MODEL, input="hi", store=True)
|
||||
|
||||
def test_background_raises(self):
|
||||
with pytest.raises(litellm.UnsupportedParamsError, match="background"):
|
||||
_capture(model=CORE_MODEL, input="hi", background=True)
|
||||
|
||||
def test_previous_response_id_raises(self):
|
||||
with pytest.raises(litellm.UnsupportedParamsError, match="previous_response_id"):
|
||||
_capture(model=CORE_MODEL, input="hi", previous_response_id="resp_1")
|
||||
|
||||
def test_drop_params_strips_all_three(self, monkeypatch: pytest.MonkeyPatch):
|
||||
monkeypatch.setattr(litellm, "drop_params", True)
|
||||
captured = _capture(
|
||||
model=CORE_MODEL,
|
||||
input="hi",
|
||||
store=True,
|
||||
background=True,
|
||||
previous_response_id="resp_1",
|
||||
)
|
||||
|
||||
assert captured["body"]["store"] is False
|
||||
assert "background" not in captured["body"]
|
||||
assert "previous_response_id" not in captured["body"]
|
||||
|
||||
def test_stateful_params_are_not_advertised(self):
|
||||
supported = _responses_config("apodex-1.1").get_supported_openai_params("apodex-1.1")
|
||||
assert "background" not in supported
|
||||
assert "previous_response_id" not in supported
|
||||
assert "max_output_tokens" in supported
|
||||
|
||||
def test_max_output_tokens_still_passes_through(self):
|
||||
captured = _capture(model=CORE_MODEL, input="hi", max_output_tokens=512)
|
||||
assert captured["body"]["max_output_tokens"] == 512
|
||||
|
||||
|
||||
class TestDeepResearchKeepsState:
|
||||
"""The agent tiers survive client disconnects, so none of this may be stripped."""
|
||||
|
||||
def test_background_passes_through(self):
|
||||
captured = _capture(model=DEEP_RESEARCH_MODEL, input="hi", background=True)
|
||||
assert captured["body"]["background"] is True
|
||||
|
||||
def test_store_and_previous_response_id_pass_through(self):
|
||||
captured = _capture(model=DEEP_RESEARCH_MODEL, input="hi", store=True, previous_response_id="resp_1")
|
||||
assert captured["body"]["store"] is True
|
||||
assert captured["body"]["previous_response_id"] == "resp_1"
|
||||
|
||||
def test_store_is_not_pinned_when_unset(self):
|
||||
captured = _capture(model=DEEP_RESEARCH_MODEL, input="hi")
|
||||
assert "store" not in captured["body"]
|
||||
|
||||
def test_stateful_params_are_advertised(self):
|
||||
supported = _responses_config("apodex-1-1-deep-research").get_supported_openai_params(
|
||||
"apodex-1-1-deep-research"
|
||||
)
|
||||
assert "background" in supported
|
||||
assert "previous_response_id" in supported
|
||||
|
|
@ -1,310 +0,0 @@
|
|||
"""
|
||||
Tests for the Apodex provider (https://platform.apodex.ai/docs).
|
||||
"""
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
import httpx
|
||||
import openai
|
||||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler
|
||||
from litellm.types.utils import LlmProviders
|
||||
from litellm.utils import ProviderConfigManager
|
||||
|
||||
REPO_ROOT = Path(__file__).parents[4]
|
||||
CORE_MODEL = "apodex/apodex-1.1"
|
||||
DEEP_RESEARCH_MODEL = "apodex/apodex-1-1-deep-research"
|
||||
|
||||
CHAT_RESPONSE = {
|
||||
"id": "chatcmpl-abc123",
|
||||
"object": "chat.completion",
|
||||
"created": 1712345678,
|
||||
"model": "apodex-1.1",
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"message": {"role": "assistant", "content": "ok", "reasoning_content": "let me think"},
|
||||
"finish_reason": "stop",
|
||||
}
|
||||
],
|
||||
"usage": {
|
||||
"prompt_tokens": 1000,
|
||||
"completion_tokens": 100,
|
||||
"total_tokens": 1100,
|
||||
"prompt_tokens_details": {"cached_tokens": 500},
|
||||
},
|
||||
}
|
||||
|
||||
STREAM_BODY = (
|
||||
b'data: {"id":"chatcmpl-abc123","object":"chat.completion.chunk","created":1,"model":"apodex-1.1",'
|
||||
b'"choices":[{"index":0,"delta":{"content":"ok"},"finish_reason":null}]}\n\n'
|
||||
b"data: [DONE]\n\n"
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _apodex_env(monkeypatch: pytest.MonkeyPatch):
|
||||
"""Resolve models against the in-repo cost map, not the published one."""
|
||||
monkeypatch.setenv("APODEX_API_KEY", "sk-apodex-test")
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
|
||||
yield
|
||||
|
||||
|
||||
def _openai_client(captured: dict, *, stream: bool = False) -> openai.OpenAI:
|
||||
def handler(request: httpx.Request) -> httpx.Response:
|
||||
captured["url"] = str(request.url)
|
||||
captured["body"] = json.loads(request.content)
|
||||
if stream:
|
||||
return httpx.Response(200, headers={"content-type": "text/event-stream"}, content=STREAM_BODY)
|
||||
return httpx.Response(200, json=CHAT_RESPONSE)
|
||||
|
||||
return openai.OpenAI(
|
||||
api_key="sk-apodex-test",
|
||||
base_url="https://api.apodex.ai/v1",
|
||||
http_client=httpx.Client(transport=httpx.MockTransport(handler)),
|
||||
)
|
||||
|
||||
|
||||
class TestApodexRegistration:
|
||||
def test_provider_enum_and_lists(self):
|
||||
assert LlmProviders.APODEX.value == "apodex"
|
||||
assert "apodex" in litellm.provider_list
|
||||
assert "apodex" in litellm.constants.openai_compatible_providers
|
||||
|
||||
def test_json_provider_config(self):
|
||||
from litellm.llms.openai_like.json_loader import JSONProviderRegistry
|
||||
|
||||
apodex = JSONProviderRegistry.get("apodex")
|
||||
assert apodex is not None
|
||||
assert apodex.base_url == "https://api.apodex.ai/v1"
|
||||
assert apodex.api_key_env == "APODEX_API_KEY"
|
||||
assert apodex.api_base_env == "APODEX_API_BASE"
|
||||
assert apodex.param_mappings["max_completion_tokens"] == "max_tokens"
|
||||
assert apodex.supported_endpoints == ["/v1/chat/completions", "/v1/responses", "/v1/messages"]
|
||||
assert JSONProviderRegistry.supports_responses_api("apodex") is True
|
||||
|
||||
def test_provider_resolution(self):
|
||||
model, provider, _, api_base = litellm.get_llm_provider(model=CORE_MODEL)
|
||||
assert (model, provider, api_base) == ("apodex-1.1", "apodex", "https://api.apodex.ai/v1")
|
||||
|
||||
def test_api_base_autodetection(self):
|
||||
_, provider, api_key, _ = litellm.get_llm_provider(model="apodex-1.1", api_base="https://api.apodex.ai/v1")
|
||||
assert provider == "apodex"
|
||||
assert api_key == "sk-apodex-test"
|
||||
|
||||
def test_explicit_api_base_and_key_win(self):
|
||||
_, provider, api_key, api_base = litellm.get_llm_provider(
|
||||
model=CORE_MODEL, api_base="https://gateway.internal/v1", api_key="sk-override"
|
||||
)
|
||||
assert (provider, api_key, api_base) == ("apodex", "sk-override", "https://gateway.internal/v1")
|
||||
|
||||
|
||||
class TestApodexStreamDefault:
|
||||
"""Apodex defaults `stream` to true, so a non-streaming call must pin it to false.
|
||||
|
||||
Regression guard: the OpenAI SDK omits `stream` when it is false, which would make
|
||||
litellm.completion() receive SSE and fail to parse it.
|
||||
"""
|
||||
|
||||
def test_chat_completion_pins_stream_false(self):
|
||||
captured: dict = {}
|
||||
response = litellm.completion(
|
||||
model=CORE_MODEL,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
client=_openai_client(captured),
|
||||
)
|
||||
|
||||
assert captured["url"] == "https://api.apodex.ai/v1/chat/completions"
|
||||
assert captured["body"]["stream"] is False
|
||||
assert captured["body"]["model"] == "apodex-1.1"
|
||||
assert response.choices[0].message.reasoning_content == "let me think"
|
||||
|
||||
def test_chat_completion_streaming_sends_stream_true(self):
|
||||
captured: dict = {}
|
||||
chunks = list(
|
||||
litellm.completion(
|
||||
model=CORE_MODEL,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
stream=True,
|
||||
client=_openai_client(captured, stream=True),
|
||||
)
|
||||
)
|
||||
|
||||
assert captured["body"]["stream"] is True
|
||||
assert chunks
|
||||
|
||||
def test_user_supplied_extra_body_is_preserved(self):
|
||||
captured: dict = {}
|
||||
litellm.completion(
|
||||
model=CORE_MODEL,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
extra_body={"mcp_servers": [{"name": "docs", "url": "https://example.com/mcp"}]},
|
||||
client=_openai_client(captured),
|
||||
)
|
||||
|
||||
assert captured["body"]["stream"] is False
|
||||
assert captured["body"]["mcp_servers"] == [{"name": "docs", "url": "https://example.com/mcp"}]
|
||||
|
||||
def test_responses_api_pins_stream_false(self):
|
||||
captured: dict = {}
|
||||
|
||||
class CapturingHandler(HTTPHandler):
|
||||
def post(self, *args, **kwargs):
|
||||
captured.update(url=kwargs.get("url"), body=kwargs.get("json"))
|
||||
raise RuntimeError("captured")
|
||||
|
||||
with pytest.raises(Exception):
|
||||
litellm.responses(model=DEEP_RESEARCH_MODEL, input="hi", client=CapturingHandler())
|
||||
|
||||
assert captured["url"] == "https://api.apodex.ai/v1/responses"
|
||||
assert captured["body"]["stream"] is False
|
||||
assert captured["body"]["model"] == "apodex-1-1-deep-research"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_responses_api_streaming_sends_stream_true(self):
|
||||
captured: dict = {}
|
||||
|
||||
class CapturingHandler(AsyncHTTPHandler):
|
||||
async def post(self, *args, **kwargs):
|
||||
captured.update(body=kwargs.get("json"))
|
||||
raise RuntimeError("captured")
|
||||
|
||||
with pytest.raises(Exception):
|
||||
await litellm.aresponses(model=DEEP_RESEARCH_MODEL, input="hi", stream=True, client=CapturingHandler())
|
||||
|
||||
assert captured["body"]["stream"] is True
|
||||
|
||||
def test_flag_is_opt_in_for_other_json_providers(self):
|
||||
from litellm.llms.openai_like.json_loader import JSONProviderRegistry
|
||||
|
||||
pinstripes = JSONProviderRegistry.get("pinstripes")
|
||||
assert pinstripes is not None
|
||||
assert "send_explicit_stream_false" not in pinstripes.special_handling
|
||||
|
||||
config = ProviderConfigManager.get_provider_chat_config(
|
||||
model="ps/glm-4.5-air", provider=LlmProviders.PINSTRIPES
|
||||
)
|
||||
params = config.map_openai_params({}, {}, "ps/glm-4.5-air", False)
|
||||
assert "stream" not in params
|
||||
assert "stream" not in (params.get("extra_body") or {})
|
||||
|
||||
|
||||
class TestApodexToolSupport:
|
||||
"""Deep research tiers reject OpenAI-style tools; core models accept them."""
|
||||
|
||||
def test_deep_research_drops_tool_params(self):
|
||||
config = ProviderConfigManager.get_provider_chat_config(
|
||||
model="apodex-1-1-deep-research", provider=LlmProviders.APODEX
|
||||
)
|
||||
supported = config.get_supported_openai_params("apodex-1-1-deep-research")
|
||||
assert "tools" not in supported
|
||||
assert "tool_choice" not in supported
|
||||
|
||||
def test_core_model_keeps_tool_params(self):
|
||||
config = ProviderConfigManager.get_provider_chat_config(model="apodex-1.1", provider=LlmProviders.APODEX)
|
||||
supported = config.get_supported_openai_params("apodex-1.1")
|
||||
assert "tools" in supported
|
||||
assert "tool_choice" in supported
|
||||
|
||||
def test_max_completion_tokens_maps_to_max_tokens(self):
|
||||
config = ProviderConfigManager.get_provider_chat_config(model="apodex-1.1", provider=LlmProviders.APODEX)
|
||||
params = config.map_openai_params({"max_completion_tokens": 512}, {}, "apodex-1.1", False)
|
||||
assert params["max_tokens"] == 512
|
||||
assert "max_completion_tokens" not in params
|
||||
|
||||
|
||||
class TestApodexAnthropicMessages:
|
||||
"""Apodex serves POST /v1/messages natively, so the payload is forwarded untranslated."""
|
||||
|
||||
def test_native_passthrough_config(self):
|
||||
config = ProviderConfigManager.get_provider_anthropic_messages_config(
|
||||
model="apodex-1.1", provider=LlmProviders.APODEX
|
||||
)
|
||||
assert config is not None
|
||||
assert type(config).__name__ == "JSONProviderAnthropicMessagesConfig"
|
||||
assert (
|
||||
config.get_complete_url(
|
||||
api_base=None, api_key=None, model="apodex-1.1", optional_params={}, litellm_params={}
|
||||
)
|
||||
== "https://api.apodex.ai/v1/messages"
|
||||
)
|
||||
|
||||
def test_headers_use_provider_api_key(self):
|
||||
config = ProviderConfigManager.get_provider_anthropic_messages_config(
|
||||
model="apodex-1.1", provider=LlmProviders.APODEX
|
||||
)
|
||||
assert config is not None
|
||||
headers, _ = config.validate_anthropic_messages_environment(
|
||||
headers={}, model="apodex-1.1", messages=[], optional_params={}, litellm_params={}
|
||||
)
|
||||
assert headers["authorization"] == "Bearer sk-apodex-test"
|
||||
assert headers["anthropic-version"] == "2023-06-01"
|
||||
|
||||
|
||||
class TestApodexModelMetadata:
|
||||
@pytest.fixture(scope="class")
|
||||
def model_cost(self) -> dict:
|
||||
with open(REPO_ROOT / "model_prices_and_context_window.json") as f:
|
||||
return json.load(f)
|
||||
|
||||
def test_core_model_pricing(self, model_cost: dict):
|
||||
info = model_cost["apodex/apodex-1.1"]
|
||||
assert info["litellm_provider"] == "apodex"
|
||||
assert info["mode"] == "chat"
|
||||
assert info["max_input_tokens"] == 262144
|
||||
assert info["input_cost_per_token"] == 3e-07
|
||||
assert info["cache_read_input_token_cost"] == 3e-08
|
||||
assert info["output_cost_per_token"] == 3e-06
|
||||
# Requests over 200K input tokens are billed at 2x across every tier
|
||||
assert info["input_cost_per_token_above_200k_tokens"] == 6e-07
|
||||
assert info["cache_read_input_token_cost_above_200k_tokens"] == 6e-08
|
||||
assert info["output_cost_per_token_above_200k_tokens"] == 6e-06
|
||||
assert info["supports_prompt_caching"] is True
|
||||
assert info["supports_function_calling"] is True
|
||||
assert info["supported_endpoints"] == ["/v1/chat/completions", "/v1/responses", "/v1/messages"]
|
||||
|
||||
def test_deep_research_model_pricing(self, model_cost: dict):
|
||||
info = model_cost["apodex/apodex-1-1-deep-research"]
|
||||
assert info["max_input_tokens"] == 131072
|
||||
assert info["max_output_tokens"] == 65536
|
||||
assert info["input_cost_per_token"] == 5e-06
|
||||
assert info["output_cost_per_token"] == 2e-05
|
||||
assert info["supports_function_calling"] is False
|
||||
assert info["supports_response_schema"] is False
|
||||
assert info["supports_prompt_caching"] is False
|
||||
assert info["supports_web_search"] is True
|
||||
assert info["supported_endpoints"] == ["/v1/chat/completions", "/v1/responses"]
|
||||
|
||||
def test_every_apodex_model_is_registered(self, model_cost: dict):
|
||||
assert {key for key in model_cost if key.startswith("apodex/")} == {
|
||||
"apodex/apodex-1.1",
|
||||
"apodex/apodex-1.1-mini",
|
||||
"apodex/apodex-1-1-deep-research",
|
||||
"apodex/apodex-1-1-deep-solve",
|
||||
"apodex/apodex-1-1-deep-discover",
|
||||
"apodex/apodex-1-0-deep-research",
|
||||
"apodex/apodex-1-0-deep-solve",
|
||||
"apodex/apodex-1-0-deep-discover",
|
||||
}
|
||||
|
||||
def test_backup_cost_map_in_sync(self, model_cost: dict):
|
||||
with open(REPO_ROOT / "litellm" / "model_prices_and_context_window_backup.json") as f:
|
||||
backup = json.load(f)
|
||||
for key in (key for key in model_cost if key.startswith("apodex/")):
|
||||
assert backup[key] == model_cost[key], f"{key} differs between main and backup cost maps"
|
||||
|
||||
def test_cost_tracks_cached_input_separately(self):
|
||||
captured: dict = {}
|
||||
response = litellm.completion(
|
||||
model=CORE_MODEL,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
client=_openai_client(captured),
|
||||
)
|
||||
|
||||
# 500 fresh input + 500 cached input + 100 output
|
||||
expected = 500 * 3e-07 + 500 * 3e-08 + 100 * 3e-06
|
||||
assert litellm.completion_cost(response, model=CORE_MODEL) == pytest.approx(expected)
|
||||
|
|
@ -245,63 +245,6 @@ class TestPinstripes:
|
|||
assert result["temperature"] == 0.7
|
||||
|
||||
|
||||
class TestTemperatureConstraints:
|
||||
"""`constraints` in providers.json clamp temperature before the request is sent."""
|
||||
|
||||
@staticmethod
|
||||
def _config(constraints: dict):
|
||||
from litellm.llms.openai_like.dynamic_config import create_config_class
|
||||
from litellm.llms.openai_like.json_loader import SimpleProviderConfig
|
||||
|
||||
provider = SimpleProviderConfig(
|
||||
"constrained",
|
||||
{
|
||||
"base_url": "https://api.constrained.test/v1",
|
||||
"api_key_env": "CONSTRAINED_API_KEY",
|
||||
"constraints": constraints,
|
||||
},
|
||||
)
|
||||
return create_config_class(provider)()
|
||||
|
||||
def test_temperature_clamped_to_max(self):
|
||||
config = self._config({"temperature_max": 1.0})
|
||||
result = config.map_openai_params({"temperature": 1.8}, {}, "some-model", False)
|
||||
assert result["temperature"] == 1.0
|
||||
|
||||
def test_temperature_clamped_to_min(self):
|
||||
config = self._config({"temperature_min": 0.1})
|
||||
result = config.map_openai_params({"temperature": 0.0}, {}, "some-model", False)
|
||||
assert result["temperature"] == 0.1
|
||||
|
||||
def test_temperature_within_range_is_untouched(self):
|
||||
config = self._config({"temperature_min": 0.1, "temperature_max": 1.0})
|
||||
result = config.map_openai_params({"temperature": 0.7}, {}, "some-model", False)
|
||||
assert result["temperature"] == 0.7
|
||||
|
||||
def test_temperature_floor_applies_only_when_n_gt_1(self):
|
||||
config = self._config({"temperature_min_with_n_gt_1": 0.3})
|
||||
|
||||
single = config.map_openai_params({"temperature": 0.0, "n": 1}, {}, "some-model", False)
|
||||
assert single["temperature"] == 0.0
|
||||
|
||||
multiple = config.map_openai_params({"temperature": 0.0, "n": 2}, {}, "some-model", False)
|
||||
assert multiple["temperature"] == 0.3
|
||||
|
||||
def test_no_constraints_leaves_temperature_alone(self):
|
||||
config = self._config({})
|
||||
result = config.map_openai_params({"temperature": 1.9}, {}, "some-model", False)
|
||||
assert result["temperature"] == 1.9
|
||||
|
||||
def test_caller_optional_params_are_not_mutated(self):
|
||||
config = self._config({"temperature_max": 1.0})
|
||||
optional_params = {"temperature": 1.8}
|
||||
result = config.map_openai_params({"max_tokens": 10}, optional_params, "some-model", False)
|
||||
|
||||
assert result["temperature"] == 1.0
|
||||
assert result["max_tokens"] == 10
|
||||
assert optional_params == {"temperature": 1.8}
|
||||
|
||||
|
||||
class TestDarkbloom:
|
||||
def test_darkbloom_json_config_exists(self):
|
||||
from litellm.llms.openai_like.json_loader import JSONProviderRegistry
|
||||
|
|
|
|||
|
|
@ -27,10 +27,10 @@
|
|||
"limit": 0
|
||||
},
|
||||
"LIT010": {
|
||||
"limit": 16707
|
||||
"limit": 16715
|
||||
},
|
||||
"LIT011": {
|
||||
"limit": 5589
|
||||
"limit": 5593
|
||||
},
|
||||
"LIT012": {
|
||||
"limit": 4519
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue