diff --git a/basedpyright-code-budget.json b/basedpyright-code-budget.json index 0109927f2c6..06010c706e3 100644 --- a/basedpyright-code-budget.json +++ b/basedpyright-code-budget.json @@ -99,7 +99,7 @@ "limit": 0 }, "reportUnknownArgumentType": { - "limit": 44774 + "limit": 44776 }, "reportUnknownLambdaType": { "limit": 113 @@ -111,7 +111,7 @@ "limit": 19967 }, "reportUnknownVariableType": { - "limit": 30879 + "limit": 30881 }, "reportUnnecessaryCast": { "limit": 117 diff --git a/litellm/__init__.py b/litellm/__init__.py index 8961de940a0..1bd171fcc7b 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -638,6 +638,7 @@ snowflake_models: Set = set() gradient_ai_models: Set = set() llama_models: Set = set() nscale_models: Set = set() +apodex_models: Set = set() nebius_models: Set = set() nebius_embedding_models: Set = set() aiml_models: Set = set() @@ -828,6 +829,8 @@ def _populate_provider_model_sets(model_cost_map: Dict) -> None: llama_models.add(key) elif value.get("litellm_provider") == "nscale": nscale_models.add(key) + elif value.get("litellm_provider") == "apodex": + apodex_models.add(key) elif value.get("litellm_provider") == "azure_ai": azure_ai_models.add(key) elif value.get("litellm_provider") == "voyage": @@ -1052,6 +1055,7 @@ model_list = list( | llama_models | featherless_ai_models | nscale_models + | apodex_models | deepgram_models | elevenlabs_models | dashscope_models @@ -1156,6 +1160,7 @@ def _build_models_by_provider() -> dict: "gradient_ai": gradient_ai_models, "meta_llama": llama_models, "nscale": nscale_models, + "apodex": apodex_models, "featherless_ai": featherless_ai_models, "deepgram": deepgram_models, "elevenlabs": elevenlabs_models, @@ -1782,6 +1787,9 @@ if TYPE_CHECKING: from .llms.perplexity.responses.transformation import ( PerplexityResponsesConfig as PerplexityResponsesConfig, ) + from .llms.apodex.responses.transformation import ( + ApodexResponsesConfig as ApodexResponsesConfig, + ) from .llms.databricks.responses.transformation import ( DatabricksResponsesAPIConfig as DatabricksResponsesAPIConfig, ) @@ -1855,6 +1863,7 @@ if TYPE_CHECKING: PerplexityChatConfig as _PerplexityChatConfig, ) from .llms.nscale.chat.transformation import NscaleConfig as _NscaleConfig + from .llms.apodex.chat.transformation import ApodexChatConfig as _ApodexChatConfig from .llms.watsonx.chat.transformation import ( IBMWatsonXChatConfig as _IBMWatsonXChatConfig, ) @@ -1890,6 +1899,7 @@ if TYPE_CHECKING: AzureOpenAIO1Config: Type[_AzureOpenAIO1Config] PerplexityChatConfig: Type[_PerplexityChatConfig] NscaleConfig: Type[_NscaleConfig] + ApodexChatConfig: Type[_ApodexChatConfig] IBMWatsonXChatConfig: Type[_IBMWatsonXChatConfig] IBMWatsonXAIConfig: Type[_IBMWatsonXAIConfig] LiteLLMProxyChatConfig: Type[_LiteLLMProxyChatConfig] diff --git a/litellm/_lazy_imports_registry.py b/litellm/_lazy_imports_registry.py index 89c72acc06d..c8e0024ab44 100644 --- a/litellm/_lazy_imports_registry.py +++ b/litellm/_lazy_imports_registry.py @@ -238,6 +238,7 @@ LLM_CONFIG_NAMES: Final = ( "HostedVLLMResponsesAPIConfig", "VolcEngineResponsesAPIConfig", "PerplexityResponsesConfig", + "ApodexResponsesConfig", "DatabricksResponsesAPIConfig", "OpenRouterResponsesAPIConfig", "BedrockMantleResponsesAPIConfig", @@ -291,6 +292,7 @@ LLM_CONFIG_NAMES: Final = ( "LmStudioEmbeddingConfig", "NscaleConfig", "PerplexityChatConfig", + "ApodexChatConfig", "AzureOpenAIO1Config", "IBMWatsonXAIConfig", "IBMWatsonXChatConfig", @@ -961,6 +963,10 @@ _LLM_CONFIGS_IMPORT_MAP: Final = { ".llms.perplexity.responses.transformation", "PerplexityResponsesConfig", ), + "ApodexResponsesConfig": ( + ".llms.apodex.responses.transformation", + "ApodexResponsesConfig", + ), "DatabricksResponsesAPIConfig": ( ".llms.databricks.responses.transformation", "DatabricksResponsesAPIConfig", @@ -1110,6 +1116,7 @@ _LLM_CONFIGS_IMPORT_MAP: Final = { ".llms.perplexity.chat.transformation", "PerplexityChatConfig", ), + "ApodexChatConfig": (".llms.apodex.chat.transformation", "ApodexChatConfig"), "AzureOpenAIO1Config": ( ".llms.azure.chat.o_series_transformation", "AzureOpenAIO1Config", diff --git a/litellm/constants.py b/litellm/constants.py index 7c99b0c9c46..0736603ab4c 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -824,7 +824,7 @@ openai_compatible_providers: Final[list] = [ "pinstripes", # Pinstripes - JSON-configured provider "darkbloom", "meta", # Meta Model API (Muse Spark) - JSON-configured provider - "apodex", # Apodex - JSON-configured provider + "apodex", ] openai_text_completion_compatible_providers: Final[list] = [ # providers that support `/v1/completions` "together_ai", diff --git a/litellm/litellm_core_utils/get_llm_provider_logic.py b/litellm/litellm_core_utils/get_llm_provider_logic.py index 9f07710007f..ac1faeda645 100644 --- a/litellm/litellm_core_utils/get_llm_provider_logic.py +++ b/litellm/litellm_core_utils/get_llm_provider_logic.py @@ -349,9 +349,10 @@ def get_llm_provider( elif endpoint == "https://api.meta.ai/v1": custom_llm_provider = "meta" dynamic_api_key = get_secret_str("META_API_KEY") - elif endpoint == "https://api.apodex.ai/v1": - custom_llm_provider = "apodex" - dynamic_api_key = get_secret_str("APODEX_API_KEY") + elif endpoint == litellm.ApodexChatConfig.API_BASE_URL: + custom_llm_provider = "apodex" # rebind-ok: dispatch chain resolves in place + # rebind-ok: dispatch chain resolves in place + dynamic_api_key = litellm.ApodexChatConfig.get_api_key() if api_base is not None and not isinstance(api_base, str): raise Exception(f"api base needs to be a string. api_base={api_base}") @@ -757,6 +758,9 @@ def _get_openai_compatible_provider_info( api_base, dynamic_api_key, ) = litellm.NscaleConfig()._get_openai_compatible_provider_info(api_base=api_base, api_key=api_key) + elif custom_llm_provider == "apodex": + api_base = litellm.ApodexChatConfig.get_api_base(api_base) # rebind-ok: dispatch chain resolves in place + dynamic_api_key = litellm.ApodexChatConfig.get_api_key(api_key) # rebind-ok: resolved in place elif custom_llm_provider == "heroku": ( api_base, diff --git a/litellm/llms/apodex/chat/transformation.py b/litellm/llms/apodex/chat/transformation.py new file mode 100644 index 00000000000..300e3e57991 --- /dev/null +++ b/litellm/llms/apodex/chat/transformation.py @@ -0,0 +1,119 @@ +""" +Apodex chat completions — OpenAI-compatible, with two provider quirks: + +- `stream` defaults to true upstream, so a non-streaming call has to say so + explicitly or Apodex answers with SSE that a plain call cannot parse +- the Deep Research tiers ignore sampling parameters and reject OpenAI-style + tools; only the core models take them + +Ref: https://platform.apodex.ai/docs/chat-completions +""" + +from collections.abc import Mapping +from typing import Final + +from litellm.llms.openai.chat.gpt_transformation import OpenAIGPTConfig + +from ..common_utils import ( + APODEX_API_BASE_URL, + get_apodex_api_base, + get_apodex_api_key, + is_deep_research_model, +) + +_DEEP_RESEARCH_PARAMS: Final = ( + "max_tokens", + "max_completion_tokens", + "stream", + "stream_options", + "extra_headers", + "max_retries", +) + +_CORE_PARAMS: Final = ( + *_DEEP_RESEARCH_PARAMS, + "temperature", + "top_p", + "stop", + "seed", + "n", + "tools", + "tool_choice", + "function_call", + "functions", + "parallel_tool_calls", +) + + +class ApodexChatConfig(OpenAIGPTConfig): + """ + Reference: https://platform.apodex.ai/docs + API Key: APODEX_API_KEY + Default API Base: https://api.apodex.ai/v1 + """ + + API_BASE_URL = APODEX_API_BASE_URL + + @property + def custom_llm_provider(self) -> str | None: + return "apodex" + + @staticmethod + def get_api_key(api_key: str | None = None) -> str | None: + return get_apodex_api_key(api_key) + + @staticmethod + def get_api_base(api_base: str | None = None) -> str | None: + return get_apodex_api_base(api_base) + + def _get_openai_compatible_provider_info( + self, api_base: str | None, api_key: str | None + ) -> tuple[str | None, str | None]: + return get_apodex_api_base(api_base), get_apodex_api_key(api_key) + + def get_supported_openai_params(self, model: str) -> list: # mutable-ok: matches the base-class signature + supported: Final = _DEEP_RESEARCH_PARAMS if is_deep_research_model(model) else _CORE_PARAMS + return list(supported) # mutable-ok: matches the base-class signature + + def map_openai_params( + self, + non_default_params: dict, # mutable-ok: matches the base-class signature + optional_params: dict, # mutable-ok: matches the base-class signature + model: str, + drop_params: bool, + ) -> dict: # mutable-ok: matches the base-class signature + mapped: Final = super().map_openai_params( + non_default_params=non_default_params, + optional_params=optional_params, + model=model, + drop_params=drop_params, + ) + + # Apodex documents max_tokens only. + renamed: Final = ( + mapped + if "max_completion_tokens" not in mapped + else { # mutable-ok: JSON request body + **{ # mutable-ok: JSON request body + key: value for key, value in mapped.items() if key != "max_completion_tokens" + }, + "max_tokens": mapped["max_completion_tokens"], + } + ) + + if renamed.get("stream"): + return renamed + + # The OpenAI SDK drops `stream` from the body when it is false, which would + # leave Apodex on its streaming default. extra_body is merged into the + # request body by the SDK, so it survives that drop. + requested_extra_body: Final = renamed.get("extra_body") + extra_body: Final = ( + requested_extra_body + if isinstance(requested_extra_body, Mapping) + else {} # mutable-ok: JSON request body + ) + return { # mutable-ok: JSON request body + **renamed, + "extra_body": {"stream": False, **extra_body}, # mutable-ok: JSON request body + } diff --git a/litellm/llms/apodex/common_utils.py b/litellm/llms/apodex/common_utils.py new file mode 100644 index 00000000000..ad29673ff4b --- /dev/null +++ b/litellm/llms/apodex/common_utils.py @@ -0,0 +1,36 @@ +""" +Shared helpers for the Apodex provider. + +Apodex serves two model families on one base URL, and the model id picks which +contract applies. Core models (apodex-1.1, apodex-1.1-mini) are plain inference +with native sampling parameters. The Deep Research tiers run an agent that +plans, searches and iterates, so they ignore sampling parameters, reject +OpenAI-style tools, and keep server-side state. + +Ref: https://platform.apodex.ai/docs/models +""" + +from typing import Final + +from litellm.secret_managers.main import get_secret_str + +APODEX_API_BASE_URL: Final = "https://api.apodex.ai/v1" + +_DEEP_RESEARCH_MARKER: Final = "-deep-" + + +def strip_provider_prefix(model: str) -> str: + return model.rpartition("/")[2] + + +def is_deep_research_model(model: str) -> bool: + """True for the Deep Research / Solve / Discover tiers, e.g. apodex-1-1-deep-solve.""" + return _DEEP_RESEARCH_MARKER in strip_provider_prefix(model) + + +def get_apodex_api_key(api_key: str | None = None) -> str | None: + return api_key or get_secret_str("APODEX_API_KEY") + + +def get_apodex_api_base(api_base: str | None = None) -> str: + return api_base or get_secret_str("APODEX_API_BASE") or APODEX_API_BASE_URL diff --git a/litellm/llms/apodex/messages/transformation.py b/litellm/llms/apodex/messages/transformation.py new file mode 100644 index 00000000000..7788b0b5eb7 --- /dev/null +++ b/litellm/llms/apodex/messages/transformation.py @@ -0,0 +1,52 @@ +""" +Apodex Anthropic Messages — native passthrough for the core models only. + +Apodex implements the Anthropic protocol itself at POST /v1/messages and serves +the core models there, so the payload is forwarded untranslated and +Anthropic-only features such as `thinking` and `cache_control` survive. The Deep +Research tiers are not served on that path, so `ProviderConfigManager` hands back +no config for them and they fall back to LiteLLM's Anthropic-to-chat-completions +translation. + +Ref: https://platform.apodex.ai/docs/anthropic-messages +""" + +from litellm.llms.openai_like.messages.transformation import ( + OpenAILikeAnthropicMessagesConfig, +) + +from ..common_utils import get_apodex_api_base, get_apodex_api_key + + +class ApodexAnthropicMessagesConfig(OpenAILikeAnthropicMessagesConfig): + @property + def custom_llm_provider(self) -> str | None: + return "apodex" + + def should_strip_billing_metadata(self) -> bool: + return True + + def validate_anthropic_messages_environment( + self, + headers: dict[str, str], # mutable-ok: matches the base-class signature + model: str, + messages: list[object], # mutable-ok: matches the base-class signature + optional_params: dict, # mutable-ok: matches the base-class signature + litellm_params: dict, # mutable-ok: matches the base-class signature + api_key: str | None = None, + api_base: str | None = None, + ) -> tuple[dict[str, str], str | None]: # mutable-ok: matches the base-class signature + """Fill in the Apodex credentials and base URL. + + The returned api_base is what the handler hands to get_complete_url, so + resolving it here is enough to reach the native endpoint. + """ + return super().validate_anthropic_messages_environment( + headers=headers, + model=model, + messages=messages, + optional_params=optional_params, + litellm_params=litellm_params, + api_key=get_apodex_api_key(api_key), + api_base=get_apodex_api_base(api_base), + ) diff --git a/litellm/llms/apodex/responses/transformation.py b/litellm/llms/apodex/responses/transformation.py new file mode 100644 index 00000000000..04ca2e55c74 --- /dev/null +++ b/litellm/llms/apodex/responses/transformation.py @@ -0,0 +1,100 @@ +""" +Apodex Responses API — OpenAI-compatible, with a model-aware parameter contract. + +Apodex serves /v1/responses for both model families but they accept different +subsets, so the restrictions here are keyed off the model rather than applied +provider-wide: + +- core models are a stateless subset: `store` is forced to false, and + `previous_response_id` or `background` come back as HTTP 400 +- the Deep Research tiers keep server-side state, so they take all three +- both default `stream` to true, so a non-streaming call has to say so + +Ref: https://platform.apodex.ai/docs/responses-api + https://platform.apodex.ai/docs/models +""" + +from collections.abc import Mapping +from typing import Final + +import litellm +from litellm.llms.openai.responses.transformation import OpenAIResponsesAPIConfig +from litellm.types.llms.openai import ResponsesAPIOptionalRequestParams +from litellm.types.router import GenericLiteLLMParams +from litellm.types.utils import LlmProviders + +from ..common_utils import get_apodex_api_key, is_deep_research_model + +# Rejected by the core models with HTTP 400: there is no server-side conversation +# to resume and requests are always executed inline. +_STATEFUL_PARAMS: Final = ("previous_response_id", "background") + + +class ApodexResponsesConfig(OpenAIResponsesAPIConfig): + @property + def custom_llm_provider(self) -> LlmProviders: + return LlmProviders.APODEX + + def validate_environment( + self, + headers: dict, # mutable-ok: matches the base-class signature + model: str, + litellm_params: GenericLiteLLMParams | None, + ) -> dict: # mutable-ok: matches the base-class signature + """Resolve the Apodex key rather than inheriting OpenAI's OPENAI_API_KEY fallback, + which would otherwise forward an unrelated OpenAI key to Apodex.""" + resolved_params: Final = litellm_params or GenericLiteLLMParams() + api_key: Final = get_apodex_api_key(resolved_params.api_key) + if api_key is None: + return headers + return { # mutable-ok: matches the base-class signature + **headers, + "Content-Type": "application/json", + "Authorization": f"Bearer {api_key}", + } + + def get_supported_openai_params(self, model: str) -> list: # mutable-ok: matches the base-class signature + inherited: Final = super().get_supported_openai_params(model) + if is_deep_research_model(model): + return inherited + return [ # mutable-ok: matches the base-class signature + param for param in inherited if param not in _STATEFUL_PARAMS + ] + + def map_openai_params( + self, + response_api_optional_params: ResponsesAPIOptionalRequestParams, + model: str, + drop_params: bool, + ) -> dict: # mutable-ok: matches the base-class signature + mapped: Final = super().map_openai_params( + response_api_optional_params=response_api_optional_params, + model=model, + drop_params=drop_params, + ) + + stateless: Final = ( + mapped + if is_deep_research_model(model) + else self._enforce_stateless(mapped, model=model, drop_params=drop_params) + ) + + if stateless.get("stream"): + return {**stateless} # mutable-ok: JSON request body + return {**stateless, "stream": False} # mutable-ok: JSON request body + + @staticmethod + def _enforce_stateless(params: Mapping[str, object], model: str, drop_params: bool) -> Mapping[str, object]: + """Core models only: drop what the stateless subset rejects and pin store to false.""" + if params.get("store") is True and not (drop_params or litellm.drop_params): + raise litellm.UnsupportedParamsError( + message=( + f"apodex model {model} does not support store=True on /v1/responses: the endpoint is a " + "stateless subset. To drop this, set `litellm.drop_params = True`" + ), + status_code=400, + ) + kept: Final = { # mutable-ok: JSON request body + key: value for key, value in params.items() if key not in _STATEFUL_PARAMS + } + return {**kept, "store": False} # mutable-ok: JSON request body diff --git a/litellm/llms/openai_like/README.md b/litellm/llms/openai_like/README.md index fc9375f2efb..e9aaafe48a1 100644 --- a/litellm/llms/openai_like/README.md +++ b/litellm/llms/openai_like/README.md @@ -59,15 +59,7 @@ That's it! The provider will be automatically loaded and available. // Optional: Special handling flags "special_handling": { - "convert_content_list_to_string": true, - - // Send "stream": false explicitly instead of omitting it. Needed by - // providers whose /v1/chat/completions and /v1/responses default to - // streaming, where omitting the field returns SSE to a non-streaming call - "send_explicit_stream_false": true, - - // Always send "store": false on /v1/responses - "force_store_false": true + "convert_content_list_to_string": true } } } diff --git a/litellm/llms/openai_like/dynamic_config.py b/litellm/llms/openai_like/dynamic_config.py index 52168e8bb66..19e29bcdcb2 100644 --- a/litellm/llms/openai_like/dynamic_config.py +++ b/litellm/llms/openai_like/dynamic_config.py @@ -2,7 +2,7 @@ Dynamic configuration class generator for JSON-based providers. """ -from collections.abc import Coroutine, Mapping +from collections.abc import Coroutine from typing import Any, Final, Literal, overload from litellm._logging import verbose_logger @@ -17,17 +17,6 @@ from litellm.types.llms.openai import AllMessageValues from .json_loader import SimpleProviderConfig -def _clamp_temperature(temperature: float, n: int, constraints: Mapping[str, float]) -> float: - capped: Final = ( - min(temperature, constraints["temperature_max"]) if "temperature_max" in constraints else temperature - ) - floored: Final = max(capped, constraints["temperature_min"]) if "temperature_min" in constraints else capped - floor_for_multiple_choices: Final = constraints.get("temperature_min_with_n_gt_1") - if n > 1 and floor_for_multiple_choices is not None: - return max(floored, floor_for_multiple_choices) - return floored - - def create_config_class(provider: SimpleProviderConfig): """Generate config class dynamically from JSON configuration""" @@ -142,36 +131,37 @@ def create_config_class(provider: SimpleProviderConfig): """Apply parameter mappings and constraints""" supported_params: Final = self.get_supported_openai_params(model) - mapped: Final = { - **optional_params, - **{ - provider.param_mappings.get(param, param): value - for param, value in non_default_params.items() - if param in provider.param_mappings or param in supported_params - }, - } - constrained: Final = ( - mapped - if "temperature" not in mapped - else { - **mapped, - "temperature": _clamp_temperature( - temperature=mapped["temperature"], - n=mapped.get("n", 1), - constraints=provider.constraints, - ), - } - ) + # Apply supported params + for param, value in non_default_params.items(): + # Check parameter mappings first + if param in provider.param_mappings: + optional_params[provider.param_mappings[param]] = value + elif param in supported_params: + optional_params[param] = value - # The OpenAI SDK omits `stream` entirely when it is false, which makes - # stream-by-default providers answer a non-streaming call with SSE. Pin it - # on the wire through extra_body, which the SDK merges into the request body. - if not provider.special_handling.get("send_explicit_stream_false") or constrained.get("stream"): - return constrained - requested_extra_body: Final = constrained.get("extra_body") - extra_body: Final[dict] = requested_extra_body if isinstance(requested_extra_body, dict) else {} - return {**constrained, "extra_body": {"stream": False, **extra_body}} + # Apply temperature constraints if present + if "temperature" in optional_params: + temp = optional_params["temperature"] + constraints: Final = provider.constraints + + # Clamp to max + if "temperature_max" in constraints: + temp = min(temp, constraints["temperature_max"]) + + # Clamp to min + if "temperature_min" in constraints: + temp = max(temp, constraints["temperature_min"]) + + # Special case: temperature_min_with_n_gt_1 + if "temperature_min_with_n_gt_1" in constraints: + n: Final = optional_params.get("n", 1) + if n > 1 and temp < constraints["temperature_min_with_n_gt_1"]: + temp = constraints["temperature_min_with_n_gt_1"] + + optional_params["temperature"] = temp + + return optional_params @property def custom_llm_provider(self) -> str | None: @@ -242,8 +232,6 @@ def create_responses_config_class(provider: SimpleProviderConfig): ) -> dict: if provider.special_handling.get("force_store_false"): response_api_optional_request_params["store"] = False - if provider.special_handling.get("send_explicit_stream_false"): - response_api_optional_request_params.setdefault("stream", False) return super().transform_responses_api_request( model=model, input=input, diff --git a/litellm/llms/openai_like/providers.json b/litellm/llms/openai_like/providers.json index 18cf49e6ccc..164100d4194 100644 --- a/litellm/llms/openai_like/providers.json +++ b/litellm/llms/openai_like/providers.json @@ -175,18 +175,6 @@ "base_class": "openai_gpt", "supported_endpoints": ["/v1/chat/completions", "/v1/responses", "/v1/messages"] }, - "apodex": { - "base_url": "https://api.apodex.ai/v1", - "api_key_env": "APODEX_API_KEY", - "api_base_env": "APODEX_API_BASE", - "param_mappings": { - "max_completion_tokens": "max_tokens" - }, - "supported_endpoints": ["/v1/chat/completions", "/v1/responses", "/v1/messages"], - "special_handling": { - "send_explicit_stream_false": true - } - }, "pinstripes": { "base_url": "https://pinstripes.io/v1", "api_key_env": "PINSTRIPES_API_KEY", diff --git a/litellm/utils.py b/litellm/utils.py index d91d3092624..af6bc5f5867 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -7928,6 +7928,7 @@ class ProviderConfigManager: ), LlmProviders.GRADIENT_AI: (lambda: litellm.GradientAIConfig(), False), LlmProviders.NSCALE: (lambda: litellm.NscaleConfig(), False), + LlmProviders.APODEX: (lambda: litellm.ApodexChatConfig(), False), LlmProviders.HEROKU: (lambda: litellm.HerokuChatConfig(), False), LlmProviders.OCI: (lambda: litellm.OCIChatConfig(), False), LlmProviders.HYPERBOLIC: (lambda: litellm.HyperbolicChatConfig(), False), @@ -8255,6 +8256,17 @@ class ProviderConfigManager: ) return DeepSeekAnthropicMessagesConfig() + elif litellm.LlmProviders.APODEX == provider: + from litellm.llms.apodex.common_utils import is_deep_research_model + from litellm.llms.apodex.messages.transformation import ( + ApodexAnthropicMessagesConfig, + ) + + # Apodex only serves the core models on its native /v1/messages path; the + # deep research tiers get no config so they fall back to translation. + if is_deep_research_model(model): + return None + return ApodexAnthropicMessagesConfig() elif litellm.LlmProviders.TENCENT == provider: from litellm.llms.tencent.messages.transformation import ( TencentAnthropicMessagesConfig, @@ -8440,6 +8452,8 @@ class ProviderConfigManager: return litellm.ManusResponsesAPIConfig() elif litellm.LlmProviders.PERPLEXITY == provider: return litellm.PerplexityResponsesConfig() + elif litellm.LlmProviders.APODEX == provider: + return litellm.ApodexResponsesConfig() elif litellm.LlmProviders.DATABRICKS == provider: # Databricks Responses API is only compatible with OpenAI GPT models if model and "gpt" in model.lower(): diff --git a/tests/test_litellm/llms/apodex/test_apodex_chat_transformation.py b/tests/test_litellm/llms/apodex/test_apodex_chat_transformation.py new file mode 100644 index 00000000000..61fccf2483a --- /dev/null +++ b/tests/test_litellm/llms/apodex/test_apodex_chat_transformation.py @@ -0,0 +1,209 @@ +""" +Apodex chat completions transformation. +""" + +import json + +import httpx +import openai +import pytest + +import litellm +from litellm.types.utils import LlmProviders +from litellm.utils import ProviderConfigManager + +CORE_MODEL = "apodex/apodex-1.1" +DEEP_RESEARCH_MODEL = "apodex/apodex-1-1-deep-research" + +CHAT_RESPONSE = { + "id": "chatcmpl-abc123", + "object": "chat.completion", + "created": 1712345678, + "model": "apodex-1.1", + "choices": [ + { + "index": 0, + "message": {"role": "assistant", "content": "ok", "reasoning_content": "let me think"}, + "finish_reason": "stop", + } + ], + "usage": { + "prompt_tokens": 1000, + "completion_tokens": 100, + "total_tokens": 1100, + "prompt_tokens_details": {"cached_tokens": 500}, + }, +} + +STREAM_BODY = ( + b'data: {"id":"chatcmpl-abc123","object":"chat.completion.chunk","created":1,"model":"apodex-1.1",' + b'"choices":[{"index":0,"delta":{"content":"ok"},"finish_reason":null}]}\n\n' + b"data: [DONE]\n\n" +) + + +@pytest.fixture(autouse=True) +def _apodex_env(monkeypatch: pytest.MonkeyPatch): + """Resolve models against the in-repo cost map, not the published one.""" + monkeypatch.setenv("APODEX_API_KEY", "sk-apodex-test") + monkeypatch.delenv("APODEX_API_BASE", raising=False) + monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") + monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) + yield + + +def _client(captured: dict, *, stream: bool = False) -> openai.OpenAI: + def handler(request: httpx.Request) -> httpx.Response: + captured["url"] = str(request.url) + captured["body"] = json.loads(request.content) + if stream: + return httpx.Response(200, headers={"content-type": "text/event-stream"}, content=STREAM_BODY) + return httpx.Response(200, json=CHAT_RESPONSE) + + return openai.OpenAI( + api_key="sk-apodex-test", + base_url="https://api.apodex.ai/v1", + http_client=httpx.Client(transport=httpx.MockTransport(handler)), + ) + + +def _chat_config(model: str): + return ProviderConfigManager.get_provider_chat_config(model=model, provider=LlmProviders.APODEX) + + +class TestProviderResolution: + def test_prefixed_model_resolves_to_the_default_base(self): + model, provider, api_key, api_base = litellm.get_llm_provider(model=CORE_MODEL) + assert (model, provider, api_key, api_base) == ( + "apodex-1.1", + "apodex", + "sk-apodex-test", + "https://api.apodex.ai/v1", + ) + + def test_api_base_autodetection(self): + _, provider, api_key, _ = litellm.get_llm_provider(model="apodex-1.1", api_base="https://api.apodex.ai/v1") + assert provider == "apodex" + assert api_key == "sk-apodex-test" + + def test_explicit_api_base_and_key_win(self): + _, provider, api_key, api_base = litellm.get_llm_provider( + model=CORE_MODEL, api_base="https://gateway.internal/v1", api_key="sk-override" + ) + assert (provider, api_key, api_base) == ("apodex", "sk-override", "https://gateway.internal/v1") + + def test_api_base_env_override(self, monkeypatch: pytest.MonkeyPatch): + monkeypatch.setenv("APODEX_API_BASE", "https://env.apodex.test/v1") + _, _, _, api_base = litellm.get_llm_provider(model=CORE_MODEL) + assert api_base == "https://env.apodex.test/v1" + + +class TestStreamDefault: + """Apodex defaults `stream` to true, so a non-streaming call has to pin it to false. + + Regression guard: the OpenAI SDK drops `stream` from the body when it is false, + which would leave Apodex streaming SSE at a call that cannot parse it. + """ + + def test_non_streaming_call_pins_stream_false(self): + captured: dict = {} + response = litellm.completion( + model=CORE_MODEL, + messages=[{"role": "user", "content": "hi"}], + client=_client(captured), + ) + + assert captured["url"] == "https://api.apodex.ai/v1/chat/completions" + assert captured["body"]["stream"] is False + assert captured["body"]["model"] == "apodex-1.1" + assert response.choices[0].message.reasoning_content == "let me think" + + def test_streaming_call_sends_stream_true(self): + captured: dict = {} + chunks = list( + litellm.completion( + model=CORE_MODEL, + messages=[{"role": "user", "content": "hi"}], + stream=True, + client=_client(captured, stream=True), + ) + ) + + assert captured["body"]["stream"] is True + assert chunks + + def test_deep_research_models_pin_stream_too(self): + captured: dict = {} + litellm.completion( + model=DEEP_RESEARCH_MODEL, + messages=[{"role": "user", "content": "hi"}], + client=_client(captured), + ) + assert captured["body"]["stream"] is False + + def test_user_supplied_extra_body_is_preserved(self): + """Deep research tiers reach external tools through `mcp_servers` in extra_body.""" + captured: dict = {} + mcp_servers = [{"name": "docs", "url": "https://example.com/mcp"}] + litellm.completion( + model=DEEP_RESEARCH_MODEL, + messages=[{"role": "user", "content": "hi"}], + extra_body={"mcp_servers": mcp_servers}, + client=_client(captured), + ) + + assert captured["body"]["stream"] is False + assert captured["body"]["mcp_servers"] == mcp_servers + + +class TestSupportedParams: + def test_core_models_support_tools(self): + supported = _chat_config("apodex-1.1").get_supported_openai_params("apodex-1.1") + assert "tools" in supported + assert "tool_choice" in supported + assert "temperature" in supported + assert "top_p" in supported + + def test_deep_research_rejects_tools_and_sampling_params(self): + """The tiers document tools as unsupported and sampling params as ignored.""" + supported = _chat_config(DEEP_RESEARCH_MODEL).get_supported_openai_params("apodex-1-1-deep-research") + for param in ("tools", "tool_choice", "function_call", "functions", "parallel_tool_calls"): + assert param not in supported + assert "temperature" not in supported + assert "top_p" not in supported + assert "max_tokens" in supported + + def test_tools_on_a_deep_research_model_raise(self): + with pytest.raises(litellm.UnsupportedParamsError, match="tools"): + litellm.completion( + model=DEEP_RESEARCH_MODEL, + messages=[{"role": "user", "content": "hi"}], + tools=[{"type": "function", "function": {"name": "f", "parameters": {}}}], + client=_client({}), + ) + + def test_max_completion_tokens_is_renamed_to_max_tokens(self): + captured: dict = {} + litellm.completion( + model=CORE_MODEL, + messages=[{"role": "user", "content": "hi"}], + max_completion_tokens=512, + client=_client(captured), + ) + + assert captured["body"]["max_tokens"] == 512 + assert "max_completion_tokens" not in captured["body"] + + +class TestCostTracking: + def test_cached_input_is_billed_at_the_lower_rate(self): + captured: dict = {} + response = litellm.completion( + model=CORE_MODEL, + messages=[{"role": "user", "content": "hi"}], + client=_client(captured), + ) + + # 500 fresh input + 500 cached input + 100 output + expected = 500 * 3e-07 + 500 * 3e-08 + 100 * 3e-06 + assert litellm.completion_cost(response, model=CORE_MODEL) == pytest.approx(expected) diff --git a/tests/test_litellm/llms/apodex/test_apodex_common_utils.py b/tests/test_litellm/llms/apodex/test_apodex_common_utils.py new file mode 100644 index 00000000000..1bb72ca13d6 --- /dev/null +++ b/tests/test_litellm/llms/apodex/test_apodex_common_utils.py @@ -0,0 +1,136 @@ +""" +Apodex provider registration and model-family classification. +""" + +import json +from pathlib import Path + +import pytest + +import litellm +from litellm.llms.apodex.common_utils import ( + APODEX_API_BASE_URL, + get_apodex_api_base, + get_apodex_api_key, + is_deep_research_model, +) +from litellm.types.utils import LlmProviders + +REPO_ROOT = Path(__file__).parents[4] + +CORE_MODELS = ("apodex-1.1", "apodex-1.1-mini") +DEEP_RESEARCH_MODELS = ( + "apodex-1-1-deep-research", + "apodex-1-1-deep-solve", + "apodex-1-1-deep-discover", + "apodex-1-0-deep-research", + "apodex-1-0-deep-solve", + "apodex-1-0-deep-discover", +) + + +class TestModelFamily: + """The model id, not the provider, selects which Apodex contract applies.""" + + @pytest.mark.parametrize("model", CORE_MODELS) + def test_core_models_are_not_deep_research(self, model: str): + assert is_deep_research_model(model) is False + assert is_deep_research_model(f"apodex/{model}") is False + + @pytest.mark.parametrize("model", DEEP_RESEARCH_MODELS) + def test_deep_research_models_are_detected(self, model: str): + assert is_deep_research_model(model) is True + assert is_deep_research_model(f"apodex/{model}") is True + + def test_prefix_does_not_leak_into_classification(self): + """A provider prefix containing the marker must not flip a core model.""" + assert is_deep_research_model("some-deep-gateway/apodex-1.1") is False + + +class TestCredentialResolution: + def test_defaults(self, monkeypatch: pytest.MonkeyPatch): + monkeypatch.delenv("APODEX_API_BASE", raising=False) + monkeypatch.setenv("APODEX_API_KEY", "sk-env") + + assert get_apodex_api_base(None) == APODEX_API_BASE_URL + assert get_apodex_api_key(None) == "sk-env" + + def test_explicit_values_win_over_env(self, monkeypatch: pytest.MonkeyPatch): + monkeypatch.setenv("APODEX_API_BASE", "https://env.apodex.test/v1") + monkeypatch.setenv("APODEX_API_KEY", "sk-env") + + assert get_apodex_api_base("https://explicit.apodex.test/v1") == "https://explicit.apodex.test/v1" + assert get_apodex_api_key("sk-explicit") == "sk-explicit" + + def test_env_base_overrides_default(self, monkeypatch: pytest.MonkeyPatch): + monkeypatch.setenv("APODEX_API_BASE", "https://env.apodex.test/v1") + assert get_apodex_api_base(None) == "https://env.apodex.test/v1" + + +class TestRegistration: + def test_provider_enum_and_lists(self): + assert LlmProviders.APODEX.value == "apodex" + assert "apodex" in litellm.provider_list + assert "apodex" in litellm.constants.openai_compatible_providers + assert APODEX_API_BASE_URL in litellm.constants.openai_compatible_endpoints + + def test_not_registered_as_a_json_provider(self): + """Apodex needs model-aware transformations, so it must not fall into the + generic JSON path, which would shadow the Python configs in provider resolution.""" + from litellm.llms.openai_like.json_loader import JSONProviderRegistry + + assert JSONProviderRegistry.exists("apodex") is False + + def test_config_classes_resolve_from_the_lazy_registry(self): + assert litellm.ApodexChatConfig().custom_llm_provider == "apodex" + assert litellm.ApodexResponsesConfig().custom_llm_provider == LlmProviders.APODEX + + +class TestModelMetadata: + @pytest.fixture(scope="class") + def model_cost(self) -> dict: + with open(REPO_ROOT / "model_prices_and_context_window.json") as f: + return json.load(f) + + def test_every_apodex_model_is_registered(self, model_cost: dict): + assert {key for key in model_cost if key.startswith("apodex/")} == { + f"apodex/{model}" for model in (*CORE_MODELS, *DEEP_RESEARCH_MODELS) + } + + def test_core_model_pricing(self, model_cost: dict): + info = model_cost["apodex/apodex-1.1"] + assert info["litellm_provider"] == "apodex" + assert info["mode"] == "chat" + assert info["max_input_tokens"] == 262144 + assert info["input_cost_per_token"] == 3e-07 + assert info["cache_read_input_token_cost"] == 3e-08 + assert info["output_cost_per_token"] == 3e-06 + # Requests over 200K input tokens are billed at 2x across every tier + assert info["input_cost_per_token_above_200k_tokens"] == 6e-07 + assert info["cache_read_input_token_cost_above_200k_tokens"] == 6e-08 + assert info["output_cost_per_token_above_200k_tokens"] == 6e-06 + assert info["supports_prompt_caching"] is True + assert info["supports_function_calling"] is True + assert info["supported_endpoints"] == ["/v1/chat/completions", "/v1/responses", "/v1/messages"] + + def test_deep_research_model_pricing(self, model_cost: dict): + info = model_cost["apodex/apodex-1-1-deep-research"] + assert info["max_input_tokens"] == 131072 + assert info["max_output_tokens"] == 65536 + assert info["input_cost_per_token"] == 5e-06 + assert info["output_cost_per_token"] == 2e-05 + assert info["supports_function_calling"] is False + assert info["supports_response_schema"] is False + assert info["supports_prompt_caching"] is False + assert info["supports_web_search"] is True + + @pytest.mark.parametrize("model", DEEP_RESEARCH_MODELS) + def test_deep_research_models_are_not_on_the_native_messages_path(self, model_cost: dict, model: str): + """Apodex serves /v1/messages for the core models only.""" + assert "/v1/messages" not in model_cost[f"apodex/{model}"]["supported_endpoints"] + + def test_backup_cost_map_in_sync(self, model_cost: dict): + with open(REPO_ROOT / "litellm" / "model_prices_and_context_window_backup.json") as f: + backup = json.load(f) + for key in (key for key in model_cost if key.startswith("apodex/")): + assert backup[key] == model_cost[key], f"{key} differs between main and backup cost maps" diff --git a/tests/test_litellm/llms/apodex/test_apodex_messages_transformation.py b/tests/test_litellm/llms/apodex/test_apodex_messages_transformation.py new file mode 100644 index 00000000000..8be810539e3 --- /dev/null +++ b/tests/test_litellm/llms/apodex/test_apodex_messages_transformation.py @@ -0,0 +1,98 @@ +""" +Apodex Anthropic Messages transformation. + +Apodex implements the Anthropic protocol natively at POST /v1/messages, but only +serves the core models there. The Deep Research tiers must keep working on the +same route through LiteLLM's translation instead of being handed to a path that +would reject them. +""" + +import pytest + +import litellm +from litellm.types.utils import LlmProviders +from litellm.utils import ProviderConfigManager + +CORE_MODELS = ("apodex-1.1", "apodex-1.1-mini") +DEEP_RESEARCH_MODELS = ( + "apodex-1-1-deep-research", + "apodex-1-1-deep-solve", + "apodex-1-1-deep-discover", + "apodex-1-0-deep-research", + "apodex-1-0-deep-solve", + "apodex-1-0-deep-discover", +) + + +@pytest.fixture(autouse=True) +def _apodex_env(monkeypatch: pytest.MonkeyPatch): + monkeypatch.setenv("APODEX_API_KEY", "sk-apodex-test") + monkeypatch.delenv("APODEX_API_BASE", raising=False) + yield + + +def _messages_config(model: str): + return ProviderConfigManager.get_provider_anthropic_messages_config(model=model, provider=LlmProviders.APODEX) + + +def _complete_url() -> str: + """Resolve the endpoint the way the handler does: validate first, then build the URL. + + validate_anthropic_messages_environment returns the api_base the handler feeds + into get_complete_url, so the two steps have to run in that order. + """ + config = _messages_config("apodex-1.1") + assert config is not None + _, api_base = config.validate_anthropic_messages_environment( + headers={}, model="apodex-1.1", messages=[], optional_params={}, litellm_params={} + ) + return config.get_complete_url( + api_base=api_base, api_key=None, model="apodex-1.1", optional_params={}, litellm_params={} + ) + + +class TestNativePassthroughRouting: + @pytest.mark.parametrize("model", CORE_MODELS) + def test_core_models_get_the_native_config(self, model: str): + config = _messages_config(model) + assert config is not None + assert type(config).__name__ == "ApodexAnthropicMessagesConfig" + assert config.custom_llm_provider == "apodex" + + @pytest.mark.parametrize("model", DEEP_RESEARCH_MODELS) + def test_deep_research_models_fall_back_to_translation(self, model: str): + """No native config means LiteLLM translates to chat completions, which works, + instead of forwarding to a path Apodex does not serve for these tiers.""" + assert _messages_config(model) is None + + +class TestNativePassthroughRequest: + def test_url_targets_the_native_messages_path(self): + assert _complete_url() == "https://api.apodex.ai/v1/messages" + + def test_url_honours_an_api_base_override(self, monkeypatch: pytest.MonkeyPatch): + monkeypatch.setenv("APODEX_API_BASE", "https://env.apodex.test/v1") + assert _complete_url() == "https://env.apodex.test/v1/messages" + + def test_headers_use_the_provider_api_key(self): + config = _messages_config("apodex-1.1") + assert config is not None + headers, _ = config.validate_anthropic_messages_environment( + headers={}, model="apodex-1.1", messages=[], optional_params={}, litellm_params={} + ) + assert headers["authorization"] == "Bearer sk-apodex-test" + assert headers["anthropic-version"] == "2023-06-01" + assert headers["content-type"] == "application/json" + + def test_caller_supplied_auth_header_is_not_overwritten(self): + config = _messages_config("apodex-1.1") + assert config is not None + headers, _ = config.validate_anthropic_messages_environment( + headers={"x-api-key": "sk-caller"}, + model="apodex-1.1", + messages=[], + optional_params={}, + litellm_params={}, + ) + assert headers["x-api-key"] == "sk-caller" + assert "authorization" not in headers diff --git a/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py b/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py new file mode 100644 index 00000000000..d343f4553c8 --- /dev/null +++ b/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py @@ -0,0 +1,177 @@ +""" +Apodex Responses API transformation. + +The core models expose a stateless subset of /v1/responses while the Deep +Research tiers keep server-side state, so the parameter contract is keyed off +the model rather than applied provider-wide. +""" + +import pytest + +import litellm +from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler +from litellm.types.utils import LlmProviders +from litellm.utils import ProviderConfigManager + +CORE_MODEL = "apodex/apodex-1.1" +CORE_MINI_MODEL = "apodex/apodex-1.1-mini" +DEEP_RESEARCH_MODEL = "apodex/apodex-1-1-deep-research" + + +@pytest.fixture(autouse=True) +def _apodex_env(monkeypatch: pytest.MonkeyPatch): + monkeypatch.setenv("APODEX_API_KEY", "sk-apodex-test") + monkeypatch.delenv("APODEX_API_BASE", raising=False) + monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") + monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) + monkeypatch.setattr(litellm, "drop_params", False) + yield + + +_SENTINEL = "apodex-request-captured" + + +def _capture(**kwargs) -> dict: + """Run litellm.responses() and return the request it would have sent. + + Validation errors raised before the request is built propagate to the caller. + """ + captured: dict = {} + + class CapturingHandler(HTTPHandler): + def post(self, *args, **post_kwargs): + captured.update(url=post_kwargs.get("url"), body=post_kwargs.get("json")) + raise RuntimeError(_SENTINEL) + + try: + litellm.responses(client=CapturingHandler(), **kwargs) + except Exception as exc: + if _SENTINEL not in str(exc): + raise + assert captured, "no request was sent" + return captured + + +def _responses_config(model: str): + return ProviderConfigManager.get_provider_responses_api_config(model=model, provider=LlmProviders.APODEX) + + +class TestConfigSelection: + def test_python_config_is_used_for_every_apodex_model(self): + for model in ("apodex-1.1", "apodex-1.1-mini", "apodex-1-1-deep-research"): + config = _responses_config(model) + assert type(config).__name__ == "ApodexResponsesConfig" + + def test_auth_uses_the_apodex_key(self): + config = _responses_config("apodex-1.1") + assert config.validate_environment(headers={}, model="apodex-1.1", litellm_params=None) == { + "Content-Type": "application/json", + "Authorization": "Bearer sk-apodex-test", + } + + def test_auth_does_not_fall_back_to_an_openai_key(self, monkeypatch: pytest.MonkeyPatch): + """The inherited OpenAI config would forward OPENAI_API_KEY to Apodex.""" + monkeypatch.delenv("APODEX_API_KEY", raising=False) + monkeypatch.setenv("OPENAI_API_KEY", "sk-openai-must-not-leak") + + config = _responses_config("apodex-1.1") + assert config.validate_environment(headers={}, model="apodex-1.1", litellm_params=None) == {} + + def test_request_targets_the_apodex_responses_url(self): + assert _capture(model=CORE_MODEL, input="hi")["url"] == "https://api.apodex.ai/v1/responses" + + def test_request_honours_an_api_base_override(self, monkeypatch: pytest.MonkeyPatch): + monkeypatch.setenv("APODEX_API_BASE", "https://env.apodex.test/v1") + assert _capture(model=CORE_MODEL, input="hi")["url"] == "https://env.apodex.test/v1/responses" + + +class TestStreamDefault: + def test_non_streaming_pins_stream_false(self): + captured = _capture(model=DEEP_RESEARCH_MODEL, input="hi") + assert captured["url"] == "https://api.apodex.ai/v1/responses" + assert captured["body"]["stream"] is False + + @pytest.mark.asyncio + async def test_streaming_sends_stream_true(self): + captured: dict = {} + + class CapturingHandler(AsyncHTTPHandler): + async def post(self, *args, **kwargs): + captured.update(body=kwargs.get("json")) + raise RuntimeError("captured") + + with pytest.raises(Exception, match="captured"): + await litellm.aresponses( + model=DEEP_RESEARCH_MODEL, input="hi", stream=True, client=CapturingHandler() + ) + + assert captured["body"]["stream"] is True + + +class TestCoreModelStatelessSubset: + """Apodex core models reject anything that would persist state on their side.""" + + @pytest.mark.parametrize("model", [CORE_MODEL, CORE_MINI_MODEL]) + def test_store_is_pinned_false(self, model: str): + captured = _capture(model=model, input="hi") + assert captured["body"]["store"] is False + + def test_store_true_raises(self): + with pytest.raises(litellm.UnsupportedParamsError, match="store=True"): + _capture(model=CORE_MODEL, input="hi", store=True) + + def test_background_raises(self): + with pytest.raises(litellm.UnsupportedParamsError, match="background"): + _capture(model=CORE_MODEL, input="hi", background=True) + + def test_previous_response_id_raises(self): + with pytest.raises(litellm.UnsupportedParamsError, match="previous_response_id"): + _capture(model=CORE_MODEL, input="hi", previous_response_id="resp_1") + + def test_drop_params_strips_all_three(self, monkeypatch: pytest.MonkeyPatch): + monkeypatch.setattr(litellm, "drop_params", True) + captured = _capture( + model=CORE_MODEL, + input="hi", + store=True, + background=True, + previous_response_id="resp_1", + ) + + assert captured["body"]["store"] is False + assert "background" not in captured["body"] + assert "previous_response_id" not in captured["body"] + + def test_stateful_params_are_not_advertised(self): + supported = _responses_config("apodex-1.1").get_supported_openai_params("apodex-1.1") + assert "background" not in supported + assert "previous_response_id" not in supported + assert "max_output_tokens" in supported + + def test_max_output_tokens_still_passes_through(self): + captured = _capture(model=CORE_MODEL, input="hi", max_output_tokens=512) + assert captured["body"]["max_output_tokens"] == 512 + + +class TestDeepResearchKeepsState: + """The agent tiers survive client disconnects, so none of this may be stripped.""" + + def test_background_passes_through(self): + captured = _capture(model=DEEP_RESEARCH_MODEL, input="hi", background=True) + assert captured["body"]["background"] is True + + def test_store_and_previous_response_id_pass_through(self): + captured = _capture(model=DEEP_RESEARCH_MODEL, input="hi", store=True, previous_response_id="resp_1") + assert captured["body"]["store"] is True + assert captured["body"]["previous_response_id"] == "resp_1" + + def test_store_is_not_pinned_when_unset(self): + captured = _capture(model=DEEP_RESEARCH_MODEL, input="hi") + assert "store" not in captured["body"] + + def test_stateful_params_are_advertised(self): + supported = _responses_config("apodex-1-1-deep-research").get_supported_openai_params( + "apodex-1-1-deep-research" + ) + assert "background" in supported + assert "previous_response_id" in supported diff --git a/tests/test_litellm/llms/openai_like/test_apodex_provider.py b/tests/test_litellm/llms/openai_like/test_apodex_provider.py deleted file mode 100644 index cdcf9b4b50a..00000000000 --- a/tests/test_litellm/llms/openai_like/test_apodex_provider.py +++ /dev/null @@ -1,310 +0,0 @@ -""" -Tests for the Apodex provider (https://platform.apodex.ai/docs). -""" - -import json -from pathlib import Path - -import httpx -import openai -import pytest - -import litellm -from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler -from litellm.types.utils import LlmProviders -from litellm.utils import ProviderConfigManager - -REPO_ROOT = Path(__file__).parents[4] -CORE_MODEL = "apodex/apodex-1.1" -DEEP_RESEARCH_MODEL = "apodex/apodex-1-1-deep-research" - -CHAT_RESPONSE = { - "id": "chatcmpl-abc123", - "object": "chat.completion", - "created": 1712345678, - "model": "apodex-1.1", - "choices": [ - { - "index": 0, - "message": {"role": "assistant", "content": "ok", "reasoning_content": "let me think"}, - "finish_reason": "stop", - } - ], - "usage": { - "prompt_tokens": 1000, - "completion_tokens": 100, - "total_tokens": 1100, - "prompt_tokens_details": {"cached_tokens": 500}, - }, -} - -STREAM_BODY = ( - b'data: {"id":"chatcmpl-abc123","object":"chat.completion.chunk","created":1,"model":"apodex-1.1",' - b'"choices":[{"index":0,"delta":{"content":"ok"},"finish_reason":null}]}\n\n' - b"data: [DONE]\n\n" -) - - -@pytest.fixture(autouse=True) -def _apodex_env(monkeypatch: pytest.MonkeyPatch): - """Resolve models against the in-repo cost map, not the published one.""" - monkeypatch.setenv("APODEX_API_KEY", "sk-apodex-test") - monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") - monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) - yield - - -def _openai_client(captured: dict, *, stream: bool = False) -> openai.OpenAI: - def handler(request: httpx.Request) -> httpx.Response: - captured["url"] = str(request.url) - captured["body"] = json.loads(request.content) - if stream: - return httpx.Response(200, headers={"content-type": "text/event-stream"}, content=STREAM_BODY) - return httpx.Response(200, json=CHAT_RESPONSE) - - return openai.OpenAI( - api_key="sk-apodex-test", - base_url="https://api.apodex.ai/v1", - http_client=httpx.Client(transport=httpx.MockTransport(handler)), - ) - - -class TestApodexRegistration: - def test_provider_enum_and_lists(self): - assert LlmProviders.APODEX.value == "apodex" - assert "apodex" in litellm.provider_list - assert "apodex" in litellm.constants.openai_compatible_providers - - def test_json_provider_config(self): - from litellm.llms.openai_like.json_loader import JSONProviderRegistry - - apodex = JSONProviderRegistry.get("apodex") - assert apodex is not None - assert apodex.base_url == "https://api.apodex.ai/v1" - assert apodex.api_key_env == "APODEX_API_KEY" - assert apodex.api_base_env == "APODEX_API_BASE" - assert apodex.param_mappings["max_completion_tokens"] == "max_tokens" - assert apodex.supported_endpoints == ["/v1/chat/completions", "/v1/responses", "/v1/messages"] - assert JSONProviderRegistry.supports_responses_api("apodex") is True - - def test_provider_resolution(self): - model, provider, _, api_base = litellm.get_llm_provider(model=CORE_MODEL) - assert (model, provider, api_base) == ("apodex-1.1", "apodex", "https://api.apodex.ai/v1") - - def test_api_base_autodetection(self): - _, provider, api_key, _ = litellm.get_llm_provider(model="apodex-1.1", api_base="https://api.apodex.ai/v1") - assert provider == "apodex" - assert api_key == "sk-apodex-test" - - def test_explicit_api_base_and_key_win(self): - _, provider, api_key, api_base = litellm.get_llm_provider( - model=CORE_MODEL, api_base="https://gateway.internal/v1", api_key="sk-override" - ) - assert (provider, api_key, api_base) == ("apodex", "sk-override", "https://gateway.internal/v1") - - -class TestApodexStreamDefault: - """Apodex defaults `stream` to true, so a non-streaming call must pin it to false. - - Regression guard: the OpenAI SDK omits `stream` when it is false, which would make - litellm.completion() receive SSE and fail to parse it. - """ - - def test_chat_completion_pins_stream_false(self): - captured: dict = {} - response = litellm.completion( - model=CORE_MODEL, - messages=[{"role": "user", "content": "hi"}], - client=_openai_client(captured), - ) - - assert captured["url"] == "https://api.apodex.ai/v1/chat/completions" - assert captured["body"]["stream"] is False - assert captured["body"]["model"] == "apodex-1.1" - assert response.choices[0].message.reasoning_content == "let me think" - - def test_chat_completion_streaming_sends_stream_true(self): - captured: dict = {} - chunks = list( - litellm.completion( - model=CORE_MODEL, - messages=[{"role": "user", "content": "hi"}], - stream=True, - client=_openai_client(captured, stream=True), - ) - ) - - assert captured["body"]["stream"] is True - assert chunks - - def test_user_supplied_extra_body_is_preserved(self): - captured: dict = {} - litellm.completion( - model=CORE_MODEL, - messages=[{"role": "user", "content": "hi"}], - extra_body={"mcp_servers": [{"name": "docs", "url": "https://example.com/mcp"}]}, - client=_openai_client(captured), - ) - - assert captured["body"]["stream"] is False - assert captured["body"]["mcp_servers"] == [{"name": "docs", "url": "https://example.com/mcp"}] - - def test_responses_api_pins_stream_false(self): - captured: dict = {} - - class CapturingHandler(HTTPHandler): - def post(self, *args, **kwargs): - captured.update(url=kwargs.get("url"), body=kwargs.get("json")) - raise RuntimeError("captured") - - with pytest.raises(Exception): - litellm.responses(model=DEEP_RESEARCH_MODEL, input="hi", client=CapturingHandler()) - - assert captured["url"] == "https://api.apodex.ai/v1/responses" - assert captured["body"]["stream"] is False - assert captured["body"]["model"] == "apodex-1-1-deep-research" - - @pytest.mark.asyncio - async def test_responses_api_streaming_sends_stream_true(self): - captured: dict = {} - - class CapturingHandler(AsyncHTTPHandler): - async def post(self, *args, **kwargs): - captured.update(body=kwargs.get("json")) - raise RuntimeError("captured") - - with pytest.raises(Exception): - await litellm.aresponses(model=DEEP_RESEARCH_MODEL, input="hi", stream=True, client=CapturingHandler()) - - assert captured["body"]["stream"] is True - - def test_flag_is_opt_in_for_other_json_providers(self): - from litellm.llms.openai_like.json_loader import JSONProviderRegistry - - pinstripes = JSONProviderRegistry.get("pinstripes") - assert pinstripes is not None - assert "send_explicit_stream_false" not in pinstripes.special_handling - - config = ProviderConfigManager.get_provider_chat_config( - model="ps/glm-4.5-air", provider=LlmProviders.PINSTRIPES - ) - params = config.map_openai_params({}, {}, "ps/glm-4.5-air", False) - assert "stream" not in params - assert "stream" not in (params.get("extra_body") or {}) - - -class TestApodexToolSupport: - """Deep research tiers reject OpenAI-style tools; core models accept them.""" - - def test_deep_research_drops_tool_params(self): - config = ProviderConfigManager.get_provider_chat_config( - model="apodex-1-1-deep-research", provider=LlmProviders.APODEX - ) - supported = config.get_supported_openai_params("apodex-1-1-deep-research") - assert "tools" not in supported - assert "tool_choice" not in supported - - def test_core_model_keeps_tool_params(self): - config = ProviderConfigManager.get_provider_chat_config(model="apodex-1.1", provider=LlmProviders.APODEX) - supported = config.get_supported_openai_params("apodex-1.1") - assert "tools" in supported - assert "tool_choice" in supported - - def test_max_completion_tokens_maps_to_max_tokens(self): - config = ProviderConfigManager.get_provider_chat_config(model="apodex-1.1", provider=LlmProviders.APODEX) - params = config.map_openai_params({"max_completion_tokens": 512}, {}, "apodex-1.1", False) - assert params["max_tokens"] == 512 - assert "max_completion_tokens" not in params - - -class TestApodexAnthropicMessages: - """Apodex serves POST /v1/messages natively, so the payload is forwarded untranslated.""" - - def test_native_passthrough_config(self): - config = ProviderConfigManager.get_provider_anthropic_messages_config( - model="apodex-1.1", provider=LlmProviders.APODEX - ) - assert config is not None - assert type(config).__name__ == "JSONProviderAnthropicMessagesConfig" - assert ( - config.get_complete_url( - api_base=None, api_key=None, model="apodex-1.1", optional_params={}, litellm_params={} - ) - == "https://api.apodex.ai/v1/messages" - ) - - def test_headers_use_provider_api_key(self): - config = ProviderConfigManager.get_provider_anthropic_messages_config( - model="apodex-1.1", provider=LlmProviders.APODEX - ) - assert config is not None - headers, _ = config.validate_anthropic_messages_environment( - headers={}, model="apodex-1.1", messages=[], optional_params={}, litellm_params={} - ) - assert headers["authorization"] == "Bearer sk-apodex-test" - assert headers["anthropic-version"] == "2023-06-01" - - -class TestApodexModelMetadata: - @pytest.fixture(scope="class") - def model_cost(self) -> dict: - with open(REPO_ROOT / "model_prices_and_context_window.json") as f: - return json.load(f) - - def test_core_model_pricing(self, model_cost: dict): - info = model_cost["apodex/apodex-1.1"] - assert info["litellm_provider"] == "apodex" - assert info["mode"] == "chat" - assert info["max_input_tokens"] == 262144 - assert info["input_cost_per_token"] == 3e-07 - assert info["cache_read_input_token_cost"] == 3e-08 - assert info["output_cost_per_token"] == 3e-06 - # Requests over 200K input tokens are billed at 2x across every tier - assert info["input_cost_per_token_above_200k_tokens"] == 6e-07 - assert info["cache_read_input_token_cost_above_200k_tokens"] == 6e-08 - assert info["output_cost_per_token_above_200k_tokens"] == 6e-06 - assert info["supports_prompt_caching"] is True - assert info["supports_function_calling"] is True - assert info["supported_endpoints"] == ["/v1/chat/completions", "/v1/responses", "/v1/messages"] - - def test_deep_research_model_pricing(self, model_cost: dict): - info = model_cost["apodex/apodex-1-1-deep-research"] - assert info["max_input_tokens"] == 131072 - assert info["max_output_tokens"] == 65536 - assert info["input_cost_per_token"] == 5e-06 - assert info["output_cost_per_token"] == 2e-05 - assert info["supports_function_calling"] is False - assert info["supports_response_schema"] is False - assert info["supports_prompt_caching"] is False - assert info["supports_web_search"] is True - assert info["supported_endpoints"] == ["/v1/chat/completions", "/v1/responses"] - - def test_every_apodex_model_is_registered(self, model_cost: dict): - assert {key for key in model_cost if key.startswith("apodex/")} == { - "apodex/apodex-1.1", - "apodex/apodex-1.1-mini", - "apodex/apodex-1-1-deep-research", - "apodex/apodex-1-1-deep-solve", - "apodex/apodex-1-1-deep-discover", - "apodex/apodex-1-0-deep-research", - "apodex/apodex-1-0-deep-solve", - "apodex/apodex-1-0-deep-discover", - } - - def test_backup_cost_map_in_sync(self, model_cost: dict): - with open(REPO_ROOT / "litellm" / "model_prices_and_context_window_backup.json") as f: - backup = json.load(f) - for key in (key for key in model_cost if key.startswith("apodex/")): - assert backup[key] == model_cost[key], f"{key} differs between main and backup cost maps" - - def test_cost_tracks_cached_input_separately(self): - captured: dict = {} - response = litellm.completion( - model=CORE_MODEL, - messages=[{"role": "user", "content": "hi"}], - client=_openai_client(captured), - ) - - # 500 fresh input + 500 cached input + 100 output - expected = 500 * 3e-07 + 500 * 3e-08 + 100 * 3e-06 - assert litellm.completion_cost(response, model=CORE_MODEL) == pytest.approx(expected) diff --git a/tests/test_litellm/llms/openai_like/test_json_providers.py b/tests/test_litellm/llms/openai_like/test_json_providers.py index 442dd554885..c8743e1809d 100644 --- a/tests/test_litellm/llms/openai_like/test_json_providers.py +++ b/tests/test_litellm/llms/openai_like/test_json_providers.py @@ -245,63 +245,6 @@ class TestPinstripes: assert result["temperature"] == 0.7 -class TestTemperatureConstraints: - """`constraints` in providers.json clamp temperature before the request is sent.""" - - @staticmethod - def _config(constraints: dict): - from litellm.llms.openai_like.dynamic_config import create_config_class - from litellm.llms.openai_like.json_loader import SimpleProviderConfig - - provider = SimpleProviderConfig( - "constrained", - { - "base_url": "https://api.constrained.test/v1", - "api_key_env": "CONSTRAINED_API_KEY", - "constraints": constraints, - }, - ) - return create_config_class(provider)() - - def test_temperature_clamped_to_max(self): - config = self._config({"temperature_max": 1.0}) - result = config.map_openai_params({"temperature": 1.8}, {}, "some-model", False) - assert result["temperature"] == 1.0 - - def test_temperature_clamped_to_min(self): - config = self._config({"temperature_min": 0.1}) - result = config.map_openai_params({"temperature": 0.0}, {}, "some-model", False) - assert result["temperature"] == 0.1 - - def test_temperature_within_range_is_untouched(self): - config = self._config({"temperature_min": 0.1, "temperature_max": 1.0}) - result = config.map_openai_params({"temperature": 0.7}, {}, "some-model", False) - assert result["temperature"] == 0.7 - - def test_temperature_floor_applies_only_when_n_gt_1(self): - config = self._config({"temperature_min_with_n_gt_1": 0.3}) - - single = config.map_openai_params({"temperature": 0.0, "n": 1}, {}, "some-model", False) - assert single["temperature"] == 0.0 - - multiple = config.map_openai_params({"temperature": 0.0, "n": 2}, {}, "some-model", False) - assert multiple["temperature"] == 0.3 - - def test_no_constraints_leaves_temperature_alone(self): - config = self._config({}) - result = config.map_openai_params({"temperature": 1.9}, {}, "some-model", False) - assert result["temperature"] == 1.9 - - def test_caller_optional_params_are_not_mutated(self): - config = self._config({"temperature_max": 1.0}) - optional_params = {"temperature": 1.8} - result = config.map_openai_params({"max_tokens": 10}, optional_params, "some-model", False) - - assert result["temperature"] == 1.0 - assert result["max_tokens"] == 10 - assert optional_params == {"temperature": 1.8} - - class TestDarkbloom: def test_darkbloom_json_config_exists(self): from litellm.llms.openai_like.json_loader import JSONProviderRegistry diff --git a/type-discipline-budget.json b/type-discipline-budget.json index b6752ada511..8e55b1533ea 100644 --- a/type-discipline-budget.json +++ b/type-discipline-budget.json @@ -27,10 +27,10 @@ "limit": 0 }, "LIT010": { - "limit": 16707 + "limit": 16715 }, "LIT011": { - "limit": 5589 + "limit": 5593 }, "LIT012": { "limit": 4519