diff --git a/README.md b/README.md index 68aaa09ec98..5cc34b14cbe 100644 --- a/README.md +++ b/README.md @@ -279,6 +279,7 @@ curl -X POST 'http://0.0.0.0:4000/v1/chat/completions' \ | [Anthropic (`anthropic`)](https://docs.litellm.ai/docs/providers/anthropic) | ✅ | ✅ | ✅ | | | | | | ✅ | | | [Anthropic Text (`anthropic_text`)](https://docs.litellm.ai/docs/providers/anthropic) | ✅ | ✅ | ✅ | | | | | | ✅ | | | [Anyscale](https://docs.litellm.ai/docs/providers/anyscale) | ✅ | ✅ | ✅ | | | | | | | | +| [Apodex (`apodex`)](https://docs.litellm.ai/docs/providers/apodex) | ✅ | ✅ | ✅ | | | | | | | | | [AssemblyAI (`assemblyai`)](https://docs.litellm.ai/docs/pass_through/assembly_ai) | ✅ | ✅ | ✅ | | | ✅ | | | | | | [Auto Router (`auto_router`)](https://docs.litellm.ai/docs/proxy/auto_routing) | ✅ | ✅ | ✅ | | | | | | | | | [AWS - Bedrock (`bedrock`)](https://docs.litellm.ai/docs/providers/bedrock) | ✅ | ✅ | ✅ | ✅ | | | | | | ✅ | diff --git a/litellm/__init__.py b/litellm/__init__.py index eebd2dad91e..7de012a0a73 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -649,6 +649,7 @@ snowflake_models: Set = set() gradient_ai_models: Set = set() llama_models: Set = set() nscale_models: Set = set() +apodex_models: Set = set() # mutable-ok: provider model registry is populated during initialization nebius_models: Set = set() nebius_embedding_models: Set = set() aiml_models: Set = set() @@ -841,6 +842,8 @@ def _populate_provider_model_sets(model_cost_map: Dict) -> None: llama_models.add(key) elif value.get("litellm_provider") == "nscale": nscale_models.add(key) + elif value.get("litellm_provider") == "apodex": + apodex_models.add(key) elif value.get("litellm_provider") == "azure_ai": azure_ai_models.add(key) elif value.get("litellm_provider") == "voyage": @@ -1065,6 +1068,7 @@ model_list = list( | llama_models | featherless_ai_models | nscale_models + | apodex_models | deepgram_models | elevenlabs_models | dashscope_models @@ -1169,6 +1173,7 @@ def _build_models_by_provider() -> dict: "gradient_ai": gradient_ai_models, "meta_llama": llama_models, "nscale": nscale_models, + "apodex": apodex_models, "featherless_ai": featherless_ai_models, "deepgram": deepgram_models, "elevenlabs": elevenlabs_models, @@ -1798,6 +1803,9 @@ if TYPE_CHECKING: from .llms.perplexity.responses.transformation import ( PerplexityResponsesConfig as PerplexityResponsesConfig, ) + from .llms.apodex.responses.transformation import ( + ApodexResponsesConfig as ApodexResponsesConfig, + ) from .llms.databricks.responses.transformation import ( DatabricksResponsesAPIConfig as DatabricksResponsesAPIConfig, ) @@ -1874,6 +1882,7 @@ if TYPE_CHECKING: PerplexityChatConfig as _PerplexityChatConfig, ) from .llms.nscale.chat.transformation import NscaleConfig as _NscaleConfig + from .llms.apodex.chat.transformation import ApodexChatConfig as _ApodexChatConfig from .llms.watsonx.chat.transformation import ( IBMWatsonXChatConfig as _IBMWatsonXChatConfig, ) @@ -1909,6 +1918,7 @@ if TYPE_CHECKING: AzureOpenAIO1Config: Type[_AzureOpenAIO1Config] PerplexityChatConfig: Type[_PerplexityChatConfig] NscaleConfig: Type[_NscaleConfig] + ApodexChatConfig: Type[_ApodexChatConfig] IBMWatsonXChatConfig: Type[_IBMWatsonXChatConfig] IBMWatsonXAIConfig: Type[_IBMWatsonXAIConfig] LiteLLMProxyChatConfig: Type[_LiteLLMProxyChatConfig] diff --git a/litellm/_lazy_imports_registry.py b/litellm/_lazy_imports_registry.py index 1c833256598..05765aa9969 100644 --- a/litellm/_lazy_imports_registry.py +++ b/litellm/_lazy_imports_registry.py @@ -239,6 +239,7 @@ LLM_CONFIG_NAMES: Final = ( "HostedVLLMResponsesAPIConfig", "VolcEngineResponsesAPIConfig", "PerplexityResponsesConfig", + "ApodexResponsesConfig", "DatabricksResponsesAPIConfig", "OpenRouterResponsesAPIConfig", "BedrockMantleResponsesAPIConfig", @@ -293,6 +294,7 @@ LLM_CONFIG_NAMES: Final = ( "LmStudioEmbeddingConfig", "NscaleConfig", "PerplexityChatConfig", + "ApodexChatConfig", "AzureOpenAIO1Config", "IBMWatsonXAIConfig", "IBMWatsonXChatConfig", @@ -967,6 +969,10 @@ _LLM_CONFIGS_IMPORT_MAP: Final = { ".llms.perplexity.responses.transformation", "PerplexityResponsesConfig", ), + "ApodexResponsesConfig": ( + ".llms.apodex.responses.transformation", + "ApodexResponsesConfig", + ), "DatabricksResponsesAPIConfig": ( ".llms.databricks.responses.transformation", "DatabricksResponsesAPIConfig", @@ -1120,6 +1126,7 @@ _LLM_CONFIGS_IMPORT_MAP: Final = { ".llms.perplexity.chat.transformation", "PerplexityChatConfig", ), + "ApodexChatConfig": (".llms.apodex.chat.transformation", "ApodexChatConfig"), "AzureOpenAIO1Config": ( ".llms.azure.chat.o_series_transformation", "AzureOpenAIO1Config", diff --git a/litellm/constants.py b/litellm/constants.py index 23e92d26a59..0d4644e102c 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -789,6 +789,7 @@ openai_compatible_endpoints: Final[list] = [ "https://api.libertai.io/v1", "https://pinstripes.io/v1", "https://api.meta.ai/v1", + "https://api.apodex.ai/v1", "https://api.cognition.ai/v1", "https://api.scx.ai/v1", ] @@ -858,6 +859,7 @@ openai_compatible_providers: Final[list] = [ "pinstripes", # Pinstripes - JSON-configured provider "darkbloom", "meta", # Meta Model API (Muse Spark) - JSON-configured provider + "apodex", "cognition", "scx-ai", ] diff --git a/litellm/litellm_core_utils/get_llm_provider_logic.py b/litellm/litellm_core_utils/get_llm_provider_logic.py index 005e94ebe82..c8a0a8f0489 100644 --- a/litellm/litellm_core_utils/get_llm_provider_logic.py +++ b/litellm/litellm_core_utils/get_llm_provider_logic.py @@ -357,6 +357,9 @@ def get_llm_provider( elif endpoint == "https://api.meta.ai/v1": custom_llm_provider = "meta" dynamic_api_key = get_secret_str("META_API_KEY") + elif endpoint == litellm.ApodexChatConfig.API_BASE_URL: + custom_llm_provider = "apodex" # rebind-ok: dispatch chain resolves in place + dynamic_api_key = litellm.ApodexChatConfig.get_api_key() elif (json_provider := JSONProviderRegistry.get_by_base_url(endpoint)) is not None: custom_llm_provider = json_provider.slug dynamic_api_key = api_key if api_key is not None else get_secret_str(json_provider.api_key_env) @@ -765,6 +768,9 @@ def _get_openai_compatible_provider_info( api_base, dynamic_api_key, ) = litellm.NscaleConfig()._get_openai_compatible_provider_info(api_base=api_base, api_key=api_key) + elif custom_llm_provider == "apodex": + api_base = litellm.ApodexChatConfig.get_api_base(api_base) # rebind-ok: dispatch chain resolves in place + dynamic_api_key = litellm.ApodexChatConfig.get_api_key(api_key) # rebind-ok: resolved in place elif custom_llm_provider == "heroku": ( api_base, diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py index 283c706e45e..9b36da98de8 100644 --- a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py +++ b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py @@ -39,9 +39,9 @@ from ..utils import is_reasoning_auto_summary_enabled from .interceptors import get_messages_interceptors from .utils import AnthropicMessagesRequestUtils, mock_response -# Providers that are routed directly to the OpenAI Responses API instead of +# Providers that are routed directly to a Responses API instead of # going through chat/completions. -_RESPONSES_API_PROVIDERS: Final = frozenset({"openai"}) +_RESPONSES_API_PROVIDERS: Final = frozenset({"apodex", "openai"}) def _bridges_to_responses_api(model: str, custom_llm_provider: str) -> bool: @@ -567,11 +567,13 @@ def anthropic_messages_handler( anthropic_messages_provider_config = OpenAILikeAnthropicMessagesConfig() if anthropic_messages_provider_config is None: - # Route to Responses API for OpenAI / Azure, chat/completions for everything else. + # Route to a Responses API for the providers that serve one, chat/completions + # for everything else. + downstream_model: Final = model if custom_llm_provider == "apodex" else original_model _shared_kwargs: Final = dict( max_tokens=max_tokens, messages=messages, - model=original_model, + model=downstream_model, metadata=metadata, stop_sequences=stop_sequences, stream=stream, diff --git a/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py b/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py index 292d2622c7f..834d7437c43 100644 --- a/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py +++ b/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py @@ -36,6 +36,8 @@ class AnthropicResponsesStreamWrapper: self.model = model self._message_id: str = f"msg_{uuid.uuid4()}" self._current_block_index: int = -1 + self._open_block_index: int | None = None + self._open_block_type: str | None = None # Map item_id -> content_block_index so we can stop the right block later self._item_id_to_block_index: dict[str, int] = {} # Track open function_call items by item_id so we can emit tool_use start @@ -68,17 +70,49 @@ class AnthropicResponsesStreamWrapper: self._current_block_index += 1 return self._current_block_index - def _open_block(self, item_id: str | None, content_block: Mapping[str, Any]) -> int: - block_idx = self._next_block_index() - if item_id: - self._item_id_to_block_index[item_id] = block_idx + def _close_open_block(self) -> None: + if self._open_block_index is None: + return + self._chunk_queue.append( + { # mutable-ok: queued Anthropic event payload + "type": "content_block_stop", + "index": self._open_block_index, + } + ) + self._open_block_index = None + self._open_block_type = None + + def _start_block(self, block_idx: int, block_type: str, content_block: Mapping[str, object]) -> None: + self._close_open_block() self._chunk_queue.append( { "type": "content_block_start", "index": block_idx, - "content_block": content_block, + "content_block": dict(content_block), # mutable-ok: queued Anthropic event payload } ) + self._open_block_index = block_idx + self._open_block_type = block_type + + def _get_or_start_block( + self, + item_id: str | None, + block_type: str, + content_block: Mapping[str, object], + ) -> int: + mapped_index: Final = self._item_id_to_block_index.get(item_id) if item_id else None + if mapped_index is not None and mapped_index == self._open_block_index: + return mapped_index + # A resumed item whose block already closed needs a fresh one: Anthropic rejects + # a delta addressed to a stopped block. Providers that reuse one item id for a + # whole run, then interleave channels, land here. + if item_id is None and self._open_block_index is not None and self._open_block_type == block_type: + return self._open_block_index + + block_idx: Final = self._next_block_index() + if item_id: + self._item_id_to_block_index[item_id] = block_idx + self._start_block(block_idx, block_type, content_block) return block_idx def _process_event(self, event: Any) -> None: @@ -106,16 +140,26 @@ class AnthropicResponsesStreamWrapper: item_id = getattr(item, "id", None) or (item.get("id") if isinstance(item, dict) else None) if item_type == "message": - self._open_block(item_id, {"type": "text", "text": ""}) + block_idx = self._next_block_index() + if item_id: + self._item_id_to_block_index[item_id] = block_idx + self._start_block( + block_idx, + "text", + {"type": "text", "text": ""}, # mutable-ok: Anthropic content block payload + ) elif item_type == "function_call": call_id: Final = ( getattr(item, "call_id", None) or (item.get("call_id") if isinstance(item, dict) else None) or "" ) name = getattr(item, "name", None) or (item.get("name") if isinstance(item, dict) else None) or "" + block_idx = self._next_block_index() if item_id: + self._item_id_to_block_index[item_id] = block_idx self._pending_tool_ids[item_id] = call_id - self._open_block( - item_id, + self._start_block( + block_idx, + "tool_use", { "type": "tool_use", "id": call_id, @@ -129,16 +173,15 @@ class AnthropicResponsesStreamWrapper: if event_type == "response.output_text.delta": item_id = getattr(event, "item_id", None) or (event.get("item_id") if isinstance(event, dict) else None) delta = getattr(event, "delta", "") or (event.get("delta", "") if isinstance(event, dict) else "") - block_idx = self._item_id_to_block_index.get(item_id, -1) if item_id else self._current_block_index - if block_idx < 0: - # Some providers (e.g. LMStudio) skip response.output_item.added, - # so no text block is open yet; synthesize content_block_start - # instead of emitting a delta with index -1 - block_idx = self._open_block(item_id, {"type": "text", "text": ""}) + text_block_idx: Final = self._get_or_start_block( + item_id=item_id, + block_type="text", + content_block={"type": "text", "text": ""}, # mutable-ok: Anthropic content block payload + ) self._chunk_queue.append( { "type": "content_block_delta", - "index": block_idx, + "index": text_block_idx, "delta": {"type": "text_delta", "text": delta}, } ) @@ -148,18 +191,21 @@ class AnthropicResponsesStreamWrapper: if event_type == "response.reasoning_summary_text.delta": item_id = getattr(event, "item_id", None) or (event.get("item_id") if isinstance(event, dict) else None) delta = getattr(event, "delta", "") or (event.get("delta", "") if isinstance(event, dict) else "") - block_idx = self._item_id_to_block_index.get(item_id, -1) if item_id else self._current_block_index - if block_idx < 0: - if not delta: - return - block_idx = self._open_block( - item_id, - {"type": "thinking", "thinking": "", "signature": ""}, # mutable-ok: API message payload - ) + if not delta: + return + thinking_block_idx: Final = self._get_or_start_block( + item_id=item_id, + block_type="thinking", + content_block={ # mutable-ok: Anthropic content block payload + "type": "thinking", + "thinking": "", + "signature": "", + }, + ) self._chunk_queue.append( { "type": "content_block_delta", - "index": block_idx, + "index": thinking_block_idx, "delta": {"type": "thinking_delta", "thinking": delta}, } ) @@ -192,12 +238,8 @@ class AnthropicResponsesStreamWrapper: block_idx = self._item_id_to_block_index.get(item_id, -1) if item_id else self._current_block_index if block_idx < 0: return - self._chunk_queue.append( - { - "type": "content_block_stop", - "index": block_idx, - } - ) + if block_idx == self._open_block_index: + self._close_open_block() return # ---- response completed -> message_delta + message_stop ---- @@ -206,6 +248,7 @@ class AnthropicResponsesStreamWrapper: "response.failed", "response.incomplete", ): + self._close_open_block() response_obj: Final = getattr(event, "response", None) or ( event.get("response") if isinstance(event, dict) else None ) diff --git a/litellm/llms/apodex/chat/transformation.py b/litellm/llms/apodex/chat/transformation.py new file mode 100644 index 00000000000..6287aedeebb --- /dev/null +++ b/litellm/llms/apodex/chat/transformation.py @@ -0,0 +1,154 @@ +""" +Apodex chat completions — OpenAI-compatible, with two provider quirks: + +- the Deep Research tiers default `stream` to true, so a non-streaming call has + to say so explicitly or Apodex answers with SSE that a plain call cannot + parse. The core models follow OpenAI and default it to false +- the Deep Research tiers ignore sampling parameters and reject OpenAI-style + tools; only the core models take them + +Ref: https://platform.apodex.ai/docs/chat-completions + https://platform.apodex.ai/docs/models +""" + +from collections.abc import Mapping +from typing import Final + +import litellm +from litellm.llms.openai.chat.gpt_transformation import OpenAIGPTConfig +from litellm.types.llms.openai import AllMessageValues + +from ..common_utils import ( + APODEX_API_BASE_URL, + get_apodex_api_base, + get_apodex_api_key, + is_deep_research_model, + is_responses_only_model, +) + +_DEEP_RESEARCH_PARAMS: Final = ( + "max_tokens", + "max_completion_tokens", + "stream", + "stream_options", + "extra_headers", + "max_retries", +) + +_CORE_PARAMS: Final = ( + *_DEEP_RESEARCH_PARAMS, + "temperature", + "top_p", + "stop", + "seed", + "n", + "tools", + "tool_choice", + "function_call", + "functions", + "parallel_tool_calls", +) + +_PIN_NON_STREAMING: Final = "_apodex_pin_non_streaming" + + +class ApodexChatConfig(OpenAIGPTConfig): + """ + Reference: https://platform.apodex.ai/docs + API Key: APODEX_API_KEY + Default API Base: https://api.apodex.ai/v1 + """ + + API_BASE_URL = APODEX_API_BASE_URL + + @property + def custom_llm_provider(self) -> str | None: + return "apodex" + + @staticmethod + def get_api_key(api_key: str | None = None) -> str | None: + return get_apodex_api_key(api_key) + + @staticmethod + def get_api_base(api_base: str | None = None) -> str | None: + return get_apodex_api_base(api_base) + + def _get_openai_compatible_provider_info( + self, api_base: str | None, api_key: str | None + ) -> tuple[str | None, str | None]: + return get_apodex_api_base(api_base), get_apodex_api_key(api_key) + + def get_supported_openai_params(self, model: str) -> list: # mutable-ok: matches the base-class signature + supported: Final = _DEEP_RESEARCH_PARAMS if is_deep_research_model(model) else _CORE_PARAMS + return list(supported) # mutable-ok: matches the base-class signature + + def map_openai_params( + self, + non_default_params: dict, # mutable-ok: matches the base-class signature + optional_params: dict, # mutable-ok: matches the base-class signature + model: str, + drop_params: bool, + ) -> dict: # mutable-ok: matches the base-class signature + if is_responses_only_model(model): + raise litellm.BadRequestError( + message=f"apodex model {model} is only available through /v1/responses", + model=model, + llm_provider="apodex", + ) + + mapped: Final = super().map_openai_params( + non_default_params=non_default_params, + optional_params=optional_params, + model=model, + drop_params=drop_params, + ) + + renamed: Final = ( + mapped + if "max_completion_tokens" not in mapped + else { # mutable-ok: JSON request body + **{ # mutable-ok: JSON request body + key: value for key, value in mapped.items() if key != "max_completion_tokens" + }, + "max_tokens": mapped["max_completion_tokens"], + } + ) + + if renamed.get("stream"): + return renamed + + return { # mutable-ok: JSON request body + **renamed, + _PIN_NON_STREAMING: True, + } + + def transform_request( + self, + model: str, + messages: list[AllMessageValues], # mutable-ok: matches the base-class signature + optional_params: dict, # mutable-ok: matches the base-class signature + litellm_params: dict, # mutable-ok: matches the base-class signature + headers: dict, # mutable-ok: matches the base-class signature + ) -> dict: # mutable-ok: JSON request body + pin_non_streaming: Final = bool(optional_params.get(_PIN_NON_STREAMING, False)) + forwarded_params: Final = { # mutable-ok: base transformer requires a request dict + key: value for key, value in optional_params.items() if key != _PIN_NON_STREAMING + } + transformed: Final = super().transform_request( + model=model, + messages=messages, + optional_params=forwarded_params, + litellm_params=litellm_params, + headers=headers, + ) + if not pin_non_streaming: + return transformed + + requested_extra_body: Final = transformed.get("extra_body") + extra_body: Final = ( + requested_extra_body if isinstance(requested_extra_body, Mapping) else {} # mutable-ok: JSON request body + ) + return { # mutable-ok: JSON request body + **transformed, + "extra_body": {**extra_body, "stream": False}, # mutable-ok: JSON request body + } diff --git a/litellm/llms/apodex/common_utils.py b/litellm/llms/apodex/common_utils.py new file mode 100644 index 00000000000..12dcb1414b6 --- /dev/null +++ b/litellm/llms/apodex/common_utils.py @@ -0,0 +1,41 @@ +""" +Shared helpers for the Apodex provider. + +Apodex serves two model families on one base URL, and the model id picks which +contract applies. Core models (apodex-1.1, apodex-1.1-mini) are plain inference +with native sampling parameters. The Deep Research tiers run an agent that +plans, searches and iterates, so they ignore sampling parameters, reject +OpenAI-style tools, and keep server-side state. + +Ref: https://platform.apodex.ai/docs/models +""" + +from typing import Final + +from litellm.secret_managers.main import get_secret_str + +APODEX_API_BASE_URL: Final = "https://api.apodex.ai/v1" + +_DEEP_RESEARCH_MARKER: Final = "-deep-" +_RESPONSES_ONLY_MODELS: Final = frozenset({"apodex-1-1-deep-discover"}) + + +def strip_provider_prefix(model: str) -> str: + return model.rpartition("/")[2] + + +def is_deep_research_model(model: str) -> bool: + """True for the Deep Research / Solve / Discover tiers, e.g. apodex-1-1-deep-solve.""" + return _DEEP_RESEARCH_MARKER in strip_provider_prefix(model) + + +def is_responses_only_model(model: str) -> bool: + return strip_provider_prefix(model) in _RESPONSES_ONLY_MODELS + + +def get_apodex_api_key(api_key: str | None = None) -> str | None: + return api_key or get_secret_str("APODEX_API_KEY") + + +def get_apodex_api_base(api_base: str | None = None) -> str: + return api_base or get_secret_str("APODEX_API_BASE") or APODEX_API_BASE_URL diff --git a/litellm/llms/apodex/messages/transformation.py b/litellm/llms/apodex/messages/transformation.py new file mode 100644 index 00000000000..7788b0b5eb7 --- /dev/null +++ b/litellm/llms/apodex/messages/transformation.py @@ -0,0 +1,52 @@ +""" +Apodex Anthropic Messages — native passthrough for the core models only. + +Apodex implements the Anthropic protocol itself at POST /v1/messages and serves +the core models there, so the payload is forwarded untranslated and +Anthropic-only features such as `thinking` and `cache_control` survive. The Deep +Research tiers are not served on that path, so `ProviderConfigManager` hands back +no config for them and they fall back to LiteLLM's Anthropic-to-chat-completions +translation. + +Ref: https://platform.apodex.ai/docs/anthropic-messages +""" + +from litellm.llms.openai_like.messages.transformation import ( + OpenAILikeAnthropicMessagesConfig, +) + +from ..common_utils import get_apodex_api_base, get_apodex_api_key + + +class ApodexAnthropicMessagesConfig(OpenAILikeAnthropicMessagesConfig): + @property + def custom_llm_provider(self) -> str | None: + return "apodex" + + def should_strip_billing_metadata(self) -> bool: + return True + + def validate_anthropic_messages_environment( + self, + headers: dict[str, str], # mutable-ok: matches the base-class signature + model: str, + messages: list[object], # mutable-ok: matches the base-class signature + optional_params: dict, # mutable-ok: matches the base-class signature + litellm_params: dict, # mutable-ok: matches the base-class signature + api_key: str | None = None, + api_base: str | None = None, + ) -> tuple[dict[str, str], str | None]: # mutable-ok: matches the base-class signature + """Fill in the Apodex credentials and base URL. + + The returned api_base is what the handler hands to get_complete_url, so + resolving it here is enough to reach the native endpoint. + """ + return super().validate_anthropic_messages_environment( + headers=headers, + model=model, + messages=messages, + optional_params=optional_params, + litellm_params=litellm_params, + api_key=get_apodex_api_key(api_key), + api_base=get_apodex_api_base(api_base), + ) diff --git a/litellm/llms/apodex/responses/transformation.py b/litellm/llms/apodex/responses/transformation.py new file mode 100644 index 00000000000..86357e118be --- /dev/null +++ b/litellm/llms/apodex/responses/transformation.py @@ -0,0 +1,230 @@ +""" +Apodex Responses API — OpenAI-compatible, with a model-aware parameter contract. + +Apodex serves /v1/responses for both model families but they accept different +subsets, so the restrictions here are keyed off the model rather than applied +provider-wide: + +- core models are a stateless subset: `store` is forced to false, and + `previous_response_id` or `background` come back as HTTP 400 +- the Deep Research tiers keep server-side state, so they take all three +- the Deep Research tiers default `stream` to true, so a non-streaming call has + to say so; pinning it for the core models too keeps one code path + +Ref: https://platform.apodex.ai/docs/responses-api + https://platform.apodex.ai/docs/models +""" + +from __future__ import annotations + +from collections.abc import Mapping +from time import time +from types import MappingProxyType +from typing import TYPE_CHECKING, Final, NamedTuple + +import httpx +from pydantic import TypeAdapter, ValidationError + +import litellm +from litellm.llms.openai.responses.transformation import OpenAIResponsesAPIConfig +from litellm.types.llms.openai import ( + ResponsesAPIOptionalRequestParams, + ResponsesAPIResponse, + ResponsesAPIStreamingResponse, +) +from litellm.types.router import GenericLiteLLMParams +from litellm.types.utils import LlmProviders + +from ..common_utils import get_apodex_api_base, get_apodex_api_key, is_deep_research_model + +if TYPE_CHECKING: + from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj + +# Rejected by the core models with HTTP 400: there is no server-side conversation +# to resume and requests are always executed inline. +_STATEFUL_PARAMS: Final = ("previous_response_id", "background") +_CANCEL_RESPONSE_ADAPTER: Final = TypeAdapter(dict[str, object]) +_BODY_FRAMING_HEADERS: Final = frozenset({"content-encoding", "content-length"}) + +_SWARM_DELTA_EVENT: Final = "response.swarm.llm_delta" + + +class _SwarmChannel(NamedTuple): + """How one `swarm.data.channel` maps onto the OpenAI event that carries it.""" + + event_type: str + item_id_prefix: str + index_field: str + + +# A Deep Research run streams several agents. Only the reporter's `output_text` is +# the answer that lands in the final `response.completed` snapshot; the worker's +# channel-less deltas are an intermediate draft and must not be mistaken for it. +_SWARM_CHANNELS: Final = MappingProxyType( + { + "output_text": _SwarmChannel("response.output_text.delta", "msg", "content_index"), + "reasoning": _SwarmChannel("response.reasoning_summary_text.delta", "rs", "summary_index"), + } +) + + +class ApodexResponsesConfig(OpenAIResponsesAPIConfig): + @property + def custom_llm_provider(self) -> LlmProviders: + return LlmProviders.APODEX + + def validate_environment( + self, + headers: dict, # mutable-ok: matches the base-class signature + model: str, + litellm_params: GenericLiteLLMParams | None, + ) -> dict: # mutable-ok: matches the base-class signature + """Resolve the Apodex key rather than inheriting OpenAI's OPENAI_API_KEY fallback, + which would otherwise forward an unrelated OpenAI key to Apodex.""" + resolved_params: Final = litellm_params or GenericLiteLLMParams() + api_key: Final = get_apodex_api_key(resolved_params.api_key) + if api_key is None: + return headers + return { # mutable-ok: matches the base-class signature + **headers, + "Content-Type": "application/json", + "Authorization": f"Bearer {api_key}", + } + + def get_complete_url( + self, + api_base: str | None, + litellm_params: dict, # mutable-ok: matches the base-class signature + ) -> str: + resolved_base: Final = get_apodex_api_base(api_base).rstrip("/") + return f"{resolved_base}/responses" + + def transform_cancel_response_api_response( + self, + raw_response: httpx.Response, + logging_obj: LiteLLMLoggingObj, + ) -> ResponsesAPIResponse: + """Backfill the fields Apodex omits from a cancel payload but ResponsesAPIResponse requires.""" + try: + payload: Final = _CANCEL_RESPONSE_ADAPTER.validate_json(raw_response.content) + except ValidationError: + return super().transform_cancel_response_api_response( + raw_response=raw_response, + logging_obj=logging_obj, + ) + + normalized_response: Final = httpx.Response( + status_code=raw_response.status_code, + # Content-Encoding and Content-Length describe the body being replaced here; + # carrying them over makes httpx try to decompress plain JSON on read. + headers={ # mutable-ok: httpx requires mutable response headers + name: value for name, value in raw_response.headers.items() if name.lower() not in _BODY_FRAMING_HEADERS + }, + json={ # mutable-ok: httpx requires a JSON-compatible response dict + **payload, + "created_at": payload.get("created_at", int(time())), + "output": payload.get("output", []), # mutable-ok: Responses payload requires an array default + }, + ) + return super().transform_cancel_response_api_response( + raw_response=normalized_response, + logging_obj=logging_obj, + ) + + def transform_streaming_response( + self, + model: str, + parsed_chunk: dict, # mutable-ok: matches the base-class signature + logging_obj: LiteLLMLoggingObj, + ) -> ResponsesAPIStreamingResponse: + """Surface a Deep Research run's text as the OpenAI delta events callers expect. + + Observed live, not documented: the stream carries all of its text in + `response.swarm.llm_delta` and never emits `response.output_text.delta` or + any reasoning event, so without this the answer arrives only in the final + `response.completed` snapshot and the reasoning is lost. Everything this + does not recognise, the remaining `response.swarm.*` lifecycle events + included, falls through to the base class as a GenericEvent. + """ + mapped: Final = self._map_swarm_delta(parsed_chunk) + return super().transform_streaming_response( + model=model, + parsed_chunk=parsed_chunk if mapped is None else mapped, + logging_obj=logging_obj, + ) + + @staticmethod + def _map_swarm_delta( + parsed_chunk: Mapping[str, object], + ) -> dict[str, object] | None: # mutable-ok: feeds the base class's `parsed_chunk: dict` + """The OpenAI event for this swarm delta, or None to pass the chunk through.""" + if parsed_chunk.get("type") != _SWARM_DELTA_EVENT: + return None + swarm: Final = parsed_chunk.get("swarm") + data: Final = swarm.get("data") if isinstance(swarm, Mapping) else None + if not isinstance(data, Mapping): + return None + + channel_name: Final = data.get("channel") + delta: Final = data.get("delta") + if not isinstance(channel_name, str): + return None + channel: Final = _SWARM_CHANNELS.get(channel_name) + if channel is None or not isinstance(delta, str): + return None + + response_id: Final = str(parsed_chunk.get("response_id", "")) + return { # mutable-ok: JSON event payload + "type": channel.event_type, + "item_id": f"{channel.item_id_prefix}_{response_id}", + "output_index": 0, + channel.index_field: 0, + "delta": delta, + "sequence_number": parsed_chunk.get("sequence_number", 0), + } + + def get_supported_openai_params(self, model: str) -> list: # mutable-ok: matches the base-class signature + inherited: Final = super().get_supported_openai_params(model) + if is_deep_research_model(model): + return inherited + return [ # mutable-ok: matches the base-class signature + param for param in inherited if param not in _STATEFUL_PARAMS + ] + + def map_openai_params( + self, + response_api_optional_params: ResponsesAPIOptionalRequestParams, + model: str, + drop_params: bool, + ) -> dict: # mutable-ok: matches the base-class signature + mapped: Final = super().map_openai_params( + response_api_optional_params=response_api_optional_params, + model=model, + drop_params=drop_params, + ) + + stateless: Final = ( + mapped + if is_deep_research_model(model) + else self._enforce_stateless(mapped, model=model, drop_params=drop_params) + ) + + if stateless.get("stream"): + return {**stateless} # mutable-ok: JSON request body + return {**stateless, "stream": False} # mutable-ok: JSON request body + + @staticmethod + def _enforce_stateless(params: Mapping[str, object], model: str, drop_params: bool) -> Mapping[str, object]: + """Core models only: drop what the stateless subset rejects and pin store to false.""" + if params.get("store") is True and not (drop_params or litellm.drop_params): + raise litellm.UnsupportedParamsError( + message=( + f"apodex model {model} does not support store=True on /v1/responses: the endpoint is a " + "stateless subset. To drop this, set `litellm.drop_params = True`" + ), + status_code=400, + ) + kept: Final = { # mutable-ok: JSON request body + key: value for key, value in params.items() if key not in _STATEFUL_PARAMS + } + return {**kept, "store": False} # mutable-ok: JSON request body diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 9ca8d9e1bac..25b4a31c835 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -50978,6 +50978,128 @@ ], "supports_audio_output": true }, + "apodex/apodex-1.1": { + "max_tokens": 65536, + "max_input_tokens": 262144, + "max_output_tokens": 65536, + "input_cost_per_token": 3e-07, + "cache_read_input_token_cost": 3e-08, + "output_cost_per_token": 3e-06, + "input_cost_per_token_above_200k_tokens": 6e-07, + "cache_read_input_token_cost_above_200k_tokens": 6e-08, + "output_cost_per_token_above_200k_tokens": 6e-06, + "litellm_provider": "apodex", + "mode": "chat", + "source": "https://platform.apodex.ai/docs/models", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_vision": false + }, + "apodex/apodex-1.1-mini": { + "max_tokens": 65536, + "max_input_tokens": 262144, + "max_output_tokens": 65536, + "input_cost_per_token": 1e-07, + "cache_read_input_token_cost": 1e-08, + "output_cost_per_token": 1e-06, + "input_cost_per_token_above_200k_tokens": 2e-07, + "cache_read_input_token_cost_above_200k_tokens": 2e-08, + "output_cost_per_token_above_200k_tokens": 2e-06, + "litellm_provider": "apodex", + "mode": "chat", + "source": "https://platform.apodex.ai/docs/models", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_vision": false + }, + "apodex/apodex-1-1-deep-research": { + "max_tokens": 65536, + "max_input_tokens": 131072, + "max_output_tokens": 65536, + "input_cost_per_token": 5e-06, + "output_cost_per_token": 2e-05, + "litellm_provider": "apodex", + "mode": "chat", + "source": "https://platform.apodex.ai/docs/pricing", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supports_function_calling": false, + "supports_native_streaming": true, + "supports_prompt_caching": false, + "supports_reasoning": true, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_tool_choice": false, + "supports_vision": false, + "supports_web_search": true + }, + "apodex/apodex-1-1-deep-solve": { + "max_tokens": 65536, + "max_input_tokens": 131072, + "max_output_tokens": 65536, + "input_cost_per_token": 5e-06, + "output_cost_per_token": 2.5e-05, + "litellm_provider": "apodex", + "mode": "chat", + "source": "https://platform.apodex.ai/docs/pricing", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supports_function_calling": false, + "supports_native_streaming": true, + "supports_prompt_caching": false, + "supports_reasoning": true, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_tool_choice": false, + "supports_vision": false, + "supports_web_search": true + }, + "apodex/apodex-1-1-deep-discover": { + "max_tokens": 262144, + "max_input_tokens": 131072, + "max_output_tokens": 262144, + "input_cost_per_token": 1e-05, + "output_cost_per_token": 0.0001, + "litellm_provider": "apodex", + "mode": "chat", + "source": "https://platform.apodex.ai/docs/pricing", + "supported_endpoints": [ + "/v1/responses" + ], + "supports_function_calling": false, + "supports_native_streaming": true, + "supports_prompt_caching": false, + "supports_reasoning": true, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_tool_choice": false, + "supports_vision": false, + "supports_web_search": true + }, "fallback_generalizations": { "rules": [ { diff --git a/litellm/provider_endpoints_support_backup.json b/litellm/provider_endpoints_support_backup.json index 86c14fb4cd8..036a7bd97cb 100644 --- a/litellm/provider_endpoints_support_backup.json +++ b/litellm/provider_endpoints_support_backup.json @@ -177,6 +177,23 @@ "interactions": true } }, + "apodex": { + "display_name": "Apodex (`apodex`)", + "url": "https://docs.litellm.ai/docs/providers/apodex", + "endpoints": { + "chat_completions": true, + "messages": true, + "responses": true, + "embeddings": false, + "image_generations": false, + "audio_transcriptions": false, + "audio_speech": false, + "moderations": false, + "batches": false, + "rerank": false, + "a2a": false + } + }, "apertis": { "display_name": "Apertis (`apertis`)", "endpoints": { diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 73f46bd2181..be58e4fcf30 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -3809,6 +3809,7 @@ class LlmProviders(str, Enum): SCX_AI = "scx-ai" DARKBLOOM = "darkbloom" META = "meta" + APODEX = "apodex" LITELLM_AGENT = "litellm_agent" CURSOR = "cursor" BEDROCK_MANTLE = "bedrock_mantle" diff --git a/litellm/utils.py b/litellm/utils.py index 802dc151428..69814a03037 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -8067,6 +8067,7 @@ class ProviderConfigManager: ), LlmProviders.GRADIENT_AI: (lambda: litellm.GradientAIConfig(), False), LlmProviders.NSCALE: (lambda: litellm.NscaleConfig(), False), + LlmProviders.APODEX: (lambda: litellm.ApodexChatConfig(), False), LlmProviders.HEROKU: (lambda: litellm.HerokuChatConfig(), False), LlmProviders.OCI: (lambda: litellm.OCIChatConfig(), False), LlmProviders.HYPERBOLIC: (lambda: litellm.HyperbolicChatConfig(), False), @@ -8394,6 +8395,17 @@ class ProviderConfigManager: ) return DeepSeekAnthropicMessagesConfig() + elif litellm.LlmProviders.APODEX == provider: + from litellm.llms.apodex.common_utils import is_deep_research_model + from litellm.llms.apodex.messages.transformation import ( + ApodexAnthropicMessagesConfig, + ) + + # Apodex only serves the core models on its native /v1/messages path; the + # deep research tiers get no config so they fall back to translation. + if is_deep_research_model(model): + return None + return ApodexAnthropicMessagesConfig() elif litellm.LlmProviders.TENCENT == provider: from litellm.llms.tencent.messages.transformation import ( TencentAnthropicMessagesConfig, @@ -8579,6 +8591,8 @@ class ProviderConfigManager: return litellm.ManusResponsesAPIConfig() elif litellm.LlmProviders.PERPLEXITY == provider: return litellm.PerplexityResponsesConfig() + elif litellm.LlmProviders.APODEX == provider: + return litellm.ApodexResponsesConfig() elif litellm.LlmProviders.DATABRICKS == provider: # Databricks Responses API is only compatible with OpenAI GPT models if model and "gpt" in model.lower(): diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 9ca8d9e1bac..25b4a31c835 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -50978,6 +50978,128 @@ ], "supports_audio_output": true }, + "apodex/apodex-1.1": { + "max_tokens": 65536, + "max_input_tokens": 262144, + "max_output_tokens": 65536, + "input_cost_per_token": 3e-07, + "cache_read_input_token_cost": 3e-08, + "output_cost_per_token": 3e-06, + "input_cost_per_token_above_200k_tokens": 6e-07, + "cache_read_input_token_cost_above_200k_tokens": 6e-08, + "output_cost_per_token_above_200k_tokens": 6e-06, + "litellm_provider": "apodex", + "mode": "chat", + "source": "https://platform.apodex.ai/docs/models", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_vision": false + }, + "apodex/apodex-1.1-mini": { + "max_tokens": 65536, + "max_input_tokens": 262144, + "max_output_tokens": 65536, + "input_cost_per_token": 1e-07, + "cache_read_input_token_cost": 1e-08, + "output_cost_per_token": 1e-06, + "input_cost_per_token_above_200k_tokens": 2e-07, + "cache_read_input_token_cost_above_200k_tokens": 2e-08, + "output_cost_per_token_above_200k_tokens": 2e-06, + "litellm_provider": "apodex", + "mode": "chat", + "source": "https://platform.apodex.ai/docs/models", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_vision": false + }, + "apodex/apodex-1-1-deep-research": { + "max_tokens": 65536, + "max_input_tokens": 131072, + "max_output_tokens": 65536, + "input_cost_per_token": 5e-06, + "output_cost_per_token": 2e-05, + "litellm_provider": "apodex", + "mode": "chat", + "source": "https://platform.apodex.ai/docs/pricing", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supports_function_calling": false, + "supports_native_streaming": true, + "supports_prompt_caching": false, + "supports_reasoning": true, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_tool_choice": false, + "supports_vision": false, + "supports_web_search": true + }, + "apodex/apodex-1-1-deep-solve": { + "max_tokens": 65536, + "max_input_tokens": 131072, + "max_output_tokens": 65536, + "input_cost_per_token": 5e-06, + "output_cost_per_token": 2.5e-05, + "litellm_provider": "apodex", + "mode": "chat", + "source": "https://platform.apodex.ai/docs/pricing", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supports_function_calling": false, + "supports_native_streaming": true, + "supports_prompt_caching": false, + "supports_reasoning": true, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_tool_choice": false, + "supports_vision": false, + "supports_web_search": true + }, + "apodex/apodex-1-1-deep-discover": { + "max_tokens": 262144, + "max_input_tokens": 131072, + "max_output_tokens": 262144, + "input_cost_per_token": 1e-05, + "output_cost_per_token": 0.0001, + "litellm_provider": "apodex", + "mode": "chat", + "source": "https://platform.apodex.ai/docs/pricing", + "supported_endpoints": [ + "/v1/responses" + ], + "supports_function_calling": false, + "supports_native_streaming": true, + "supports_prompt_caching": false, + "supports_reasoning": true, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_tool_choice": false, + "supports_vision": false, + "supports_web_search": true + }, "fallback_generalizations": { "rules": [ { diff --git a/provider_endpoints_support.json b/provider_endpoints_support.json index 1d8d374c2c4..3057aa3d03b 100644 --- a/provider_endpoints_support.json +++ b/provider_endpoints_support.json @@ -177,6 +177,23 @@ "interactions": true } }, + "apodex": { + "display_name": "Apodex (`apodex`)", + "url": "https://docs.litellm.ai/docs/providers/apodex", + "endpoints": { + "chat_completions": true, + "messages": true, + "responses": true, + "embeddings": false, + "image_generations": false, + "audio_transcriptions": false, + "audio_speech": false, + "moderations": false, + "batches": false, + "rerank": false, + "a2a": false + } + }, "apertis": { "display_name": "Apertis (`apertis`)", "endpoints": { diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py index aebbed88c70..28339143b72 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py @@ -245,10 +245,16 @@ class TestProcessEventTextDeltaWithoutOutputItemAdded: assert chunks[1]["delta"] == {"type": "text_delta", "text": "Hel"} def test_process_event_delta_without_item_id_never_yields_negative_index(self): - chunks = _process_all([{"type": "response.output_text.delta", "delta": "Hi"}]) + chunks = _process_all( + [ + {"type": "response.output_text.delta", "delta": "Hi"}, + {"type": "response.output_text.delta", "delta": " again"}, + ] + ) assert [(c["type"], c["index"]) for c in chunks] == [ ("content_block_start", 0), ("content_block_delta", 0), + ("content_block_delta", 0), ] def test_process_event_unregistered_item_id_opens_new_text_block(self): @@ -262,9 +268,10 @@ class TestProcessEventTextDeltaWithoutOutputItemAdded: {"type": "response.output_text.delta", "item_id": "m1", "delta": "Hi"}, ] ) - assert chunks[2]["type"] == "content_block_start" - assert chunks[2]["content_block"] == {"type": "text", "text": ""} - assert [c["index"] for c in chunks[2:]] == [1, 1] + assert chunks[2] == {"type": "content_block_stop", "index": 0} + assert chunks[3]["type"] == "content_block_start" + assert chunks[3]["content_block"] == {"type": "text", "text": ""} + assert [c["index"] for c in chunks[2:]] == [0, 1, 1] def test_process_event_registered_item_id_does_not_synthesize_start(self): chunks = _process_all( @@ -282,6 +289,91 @@ class TestProcessEventTextDeltaWithoutOutputItemAdded: ] +class TestProcessEventReasoningDeltaWithoutOutputItemAdded: + def test_reasoning_opens_thinking_block_before_delta(self): + chunks = _process_all( + [ + {"type": "response.reasoning_summary_text.delta", "item_id": "rs_1", "delta": "Think "}, + {"type": "response.reasoning_summary_text.delta", "item_id": "rs_1", "delta": "carefully"}, + ] + ) + + assert [chunk["type"] for chunk in chunks] == [ + "content_block_start", + "content_block_delta", + "content_block_delta", + ] + assert chunks[0] == { + "type": "content_block_start", + "index": 0, + "content_block": {"type": "thinking", "thinking": "", "signature": ""}, + } + assert chunks[1] == { + "type": "content_block_delta", + "index": 0, + "delta": {"type": "thinking_delta", "thinking": "Think "}, + } + + def test_reasoning_and_text_get_separate_closed_blocks(self): + response = SimpleNamespace(status="completed", output=[], usage=None) + chunks = _process_all( + [ + {"type": "response.reasoning_summary_text.delta", "item_id": "rs_1", "delta": "Think"}, + {"type": "response.output_text.delta", "item_id": "msg_1", "delta": "Answer"}, + {"type": "response.completed", "response": response}, + ] + ) + + assert [(chunk["type"], chunk.get("index")) for chunk in chunks] == [ + ("content_block_start", 0), + ("content_block_delta", 0), + ("content_block_stop", 0), + ("content_block_start", 1), + ("content_block_delta", 1), + ("content_block_stop", 1), + ("message_delta", None), + ("message_stop", None), + ] + assert chunks[3]["content_block"] == {"type": "text", "text": ""} + + def test_resumed_item_id_opens_a_fresh_block(self): + """A provider that reuses one item id per run, then interleaves channels, would + otherwise address a delta to a block that has already stopped.""" + response = SimpleNamespace(status="completed", output=[], usage=None) + chunks = _process_all( + [ + {"type": "response.reasoning_summary_text.delta", "item_id": "rs_1", "delta": "Think A"}, + {"type": "response.output_text.delta", "item_id": "msg_1", "delta": "Answer A"}, + {"type": "response.reasoning_summary_text.delta", "item_id": "rs_1", "delta": "Think B"}, + {"type": "response.output_text.delta", "item_id": "msg_1", "delta": "Answer B"}, + {"type": "response.completed", "response": response}, + ] + ) + + assert [(chunk["type"], chunk.get("index")) for chunk in chunks] == [ + ("content_block_start", 0), + ("content_block_delta", 0), + ("content_block_stop", 0), + ("content_block_start", 1), + ("content_block_delta", 1), + ("content_block_stop", 1), + ("content_block_start", 2), + ("content_block_delta", 2), + ("content_block_stop", 2), + ("content_block_start", 3), + ("content_block_delta", 3), + ("content_block_stop", 3), + ("message_delta", None), + ("message_stop", None), + ] + assert [chunk["content_block"]["type"] for chunk in chunks if chunk["type"] == "content_block_start"] == [ + "thinking", + "text", + "thinking", + "text", + ] + + class TestResponseCompletedUsage: """The Anthropic ``message_delta`` usage must report cache reads/writes and exclude them from ``input_tokens``, so spend is not billed at the uncached diff --git a/tests/test_litellm/llms/apodex/test_apodex_chat_transformation.py b/tests/test_litellm/llms/apodex/test_apodex_chat_transformation.py new file mode 100644 index 00000000000..75bb3068ca3 --- /dev/null +++ b/tests/test_litellm/llms/apodex/test_apodex_chat_transformation.py @@ -0,0 +1,253 @@ +""" +Apodex chat completions transformation. +""" + +import json + +import httpx +import openai +import pytest + +import litellm +from litellm.types.utils import LlmProviders +from litellm.utils import ProviderConfigManager + +CORE_MODEL = "apodex/apodex-1.1" +DEEP_RESEARCH_MODEL = "apodex/apodex-1-1-deep-research" + +CHAT_RESPONSE = { + "id": "chatcmpl-abc123", + "object": "chat.completion", + "created": 1712345678, + "model": "apodex-1.1", + "choices": [ + { + "index": 0, + "message": {"role": "assistant", "content": "ok", "reasoning_content": "let me think"}, + "finish_reason": "stop", + } + ], + "usage": { + "prompt_tokens": 1000, + "completion_tokens": 100, + "total_tokens": 1100, + "prompt_tokens_details": {"cached_tokens": 500}, + }, +} + +STREAM_BODY = ( + b'data: {"id":"chatcmpl-abc123","object":"chat.completion.chunk","created":1,"model":"apodex-1.1",' + b'"choices":[{"index":0,"delta":{"content":"ok"},"finish_reason":null}]}\n\n' + b"data: [DONE]\n\n" +) + + +@pytest.fixture(autouse=True) +def _apodex_env(monkeypatch: pytest.MonkeyPatch): + """Resolve models against the in-repo cost map, not the published one.""" + monkeypatch.setenv("APODEX_API_KEY", "sk-apodex-test") + monkeypatch.delenv("APODEX_API_BASE", raising=False) + monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") + monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) + yield + + +def _client(captured: dict, *, stream: bool = False) -> openai.OpenAI: + def handler(request: httpx.Request) -> httpx.Response: + captured["url"] = str(request.url) + captured["body"] = json.loads(request.content) + if stream: + return httpx.Response(200, headers={"content-type": "text/event-stream"}, content=STREAM_BODY) + return httpx.Response(200, json=CHAT_RESPONSE) + + return openai.OpenAI( + api_key="sk-apodex-test", + base_url="https://api.apodex.ai/v1", + http_client=httpx.Client(transport=httpx.MockTransport(handler)), + ) + + +def _chat_config(model: str): + return ProviderConfigManager.get_provider_chat_config(model=model, provider=LlmProviders.APODEX) + + +class TestProviderResolution: + def test_openai_compatible_provider_info_uses_apodex_credentials(self): + config = _chat_config("apodex-1.1") + assert config._get_openai_compatible_provider_info("https://override.test/v1", "sk-override") == ( + "https://override.test/v1", + "sk-override", + ) + + def test_prefixed_model_resolves_to_the_default_base(self): + model, provider, api_key, api_base = litellm.get_llm_provider(model=CORE_MODEL) + assert (model, provider, api_key, api_base) == ( + "apodex-1.1", + "apodex", + "sk-apodex-test", + "https://api.apodex.ai/v1", + ) + + def test_api_base_autodetection(self): + _, provider, api_key, _ = litellm.get_llm_provider(model="apodex-1.1", api_base="https://api.apodex.ai/v1") + assert provider == "apodex" + assert api_key == "sk-apodex-test" + + def test_explicit_api_base_and_key_win(self): + _, provider, api_key, api_base = litellm.get_llm_provider( + model=CORE_MODEL, api_base="https://gateway.internal/v1", api_key="sk-override" + ) + assert (provider, api_key, api_base) == ("apodex", "sk-override", "https://gateway.internal/v1") + + def test_api_base_env_override(self, monkeypatch: pytest.MonkeyPatch): + monkeypatch.setenv("APODEX_API_BASE", "https://env.apodex.test/v1") + _, _, _, api_base = litellm.get_llm_provider(model=CORE_MODEL) + assert api_base == "https://env.apodex.test/v1" + + +class TestStreamDefault: + """The Deep Research tiers default `stream` to true, so a non-streaming call must pin it false. + + Regression guard: the OpenAI chat handler pops `stream` out of the params it + forwards, which would leave those tiers streaming SSE at a call that cannot + parse it. The core models default to false but are pinned the same way. + """ + + def test_non_streaming_call_pins_stream_false(self): + captured: dict = {} + response = litellm.completion( + model=CORE_MODEL, + messages=[{"role": "user", "content": "hi"}], + client=_client(captured), + ) + + assert captured["url"] == "https://api.apodex.ai/v1/chat/completions" + assert captured["body"]["stream"] is False + assert captured["body"]["model"] == "apodex-1.1" + assert response.choices[0].message.reasoning_content == "let me think" + + def test_streaming_call_sends_stream_true(self): + captured: dict = {} + chunks = list( + litellm.completion( + model=CORE_MODEL, + messages=[{"role": "user", "content": "hi"}], + stream=True, + client=_client(captured, stream=True), + ) + ) + + assert captured["body"]["stream"] is True + assert chunks + + def test_deep_research_models_pin_stream_too(self): + captured: dict = {} + litellm.completion( + model=DEEP_RESEARCH_MODEL, + messages=[{"role": "user", "content": "hi"}], + client=_client(captured), + ) + assert captured["body"]["stream"] is False + + def test_user_supplied_extra_body_is_preserved(self): + """Deep research tiers reach external tools through `mcp_servers` in extra_body.""" + captured: dict = {} + mcp_servers = [{"name": "docs", "url": "https://example.com/mcp"}] + litellm.completion( + model=DEEP_RESEARCH_MODEL, + messages=[{"role": "user", "content": "hi"}], + extra_body={"mcp_servers": mcp_servers}, + client=_client(captured), + ) + + assert captured["body"]["stream"] is False + assert captured["body"]["mcp_servers"] == mcp_servers + + def test_extra_body_cannot_override_non_streaming_pin(self): + captured: dict = {} + litellm.completion( + model=DEEP_RESEARCH_MODEL, + messages=[{"role": "user", "content": "hi"}], + extra_body={"stream": True}, + client=_client(captured), + ) + + assert captured["body"]["stream"] is False + + def test_transform_does_not_mutate_optional_params(self): + config = _chat_config("apodex-1.1") + optional_params = config.map_openai_params( + non_default_params={}, optional_params={}, model="apodex-1.1", drop_params=False + ) + original = optional_params.copy() + + config.transform_request( + model="apodex-1.1", + messages=[{"role": "user", "content": "hi"}], + optional_params=optional_params, + litellm_params={"custom_llm_provider": "apodex"}, + headers={}, + ) + + assert optional_params == original + + +class TestSupportedParams: + def test_responses_only_model_rejects_chat_completions(self): + with pytest.raises(litellm.BadRequestError, match="only available through /v1/responses"): + litellm.completion( + model="apodex/apodex-1-1-deep-discover", + messages=[{"role": "user", "content": "hi"}], + client=_client({}), + ) + + def test_core_models_support_tools(self): + supported = _chat_config("apodex-1.1").get_supported_openai_params("apodex-1.1") + assert "tools" in supported + assert "tool_choice" in supported + assert "temperature" in supported + assert "top_p" in supported + + def test_deep_research_rejects_tools_and_sampling_params(self): + """The tiers document tools as unsupported and sampling params as ignored.""" + supported = _chat_config(DEEP_RESEARCH_MODEL).get_supported_openai_params("apodex-1-1-deep-research") + for param in ("tools", "tool_choice", "function_call", "functions", "parallel_tool_calls"): + assert param not in supported + assert "temperature" not in supported + assert "top_p" not in supported + assert "max_tokens" in supported + + def test_tools_on_a_deep_research_model_raise(self): + with pytest.raises(litellm.UnsupportedParamsError, match="tools"): + litellm.completion( + model=DEEP_RESEARCH_MODEL, + messages=[{"role": "user", "content": "hi"}], + tools=[{"type": "function", "function": {"name": "f", "parameters": {}}}], + client=_client({}), + ) + + def test_max_completion_tokens_is_renamed_to_max_tokens(self): + captured: dict = {} + litellm.completion( + model=CORE_MODEL, + messages=[{"role": "user", "content": "hi"}], + max_completion_tokens=512, + client=_client(captured), + ) + + assert captured["body"]["max_tokens"] == 512 + assert "max_completion_tokens" not in captured["body"] + + +class TestCostTracking: + def test_cached_input_is_billed_at_the_lower_rate(self): + captured: dict = {} + response = litellm.completion( + model=CORE_MODEL, + messages=[{"role": "user", "content": "hi"}], + client=_client(captured), + ) + + # 500 fresh input + 500 cached input + 100 output + expected = 500 * 3e-07 + 500 * 3e-08 + 100 * 3e-06 + assert litellm.completion_cost(response, model=CORE_MODEL) == pytest.approx(expected) diff --git a/tests/test_litellm/llms/apodex/test_apodex_common_utils.py b/tests/test_litellm/llms/apodex/test_apodex_common_utils.py new file mode 100644 index 00000000000..da908d4d8e8 --- /dev/null +++ b/tests/test_litellm/llms/apodex/test_apodex_common_utils.py @@ -0,0 +1,153 @@ +""" +Apodex provider registration and model-family classification. +""" + +import json +from pathlib import Path + +import pytest + +import litellm +from litellm.llms.apodex.common_utils import ( + APODEX_API_BASE_URL, + get_apodex_api_base, + get_apodex_api_key, + is_deep_research_model, +) +from litellm.types.utils import LlmProviders + +REPO_ROOT = Path(__file__).parents[4] + +CORE_MODELS = ("apodex-1.1", "apodex-1.1-mini") +DEEP_RESEARCH_MODELS = ( + "apodex-1-1-deep-research", + "apodex-1-1-deep-solve", + "apodex-1-1-deep-discover", +) + + +class TestModelFamily: + """The model id, not the provider, selects which Apodex contract applies.""" + + @pytest.mark.parametrize("model", CORE_MODELS) + def test_core_models_are_not_deep_research(self, model: str): + assert is_deep_research_model(model) is False + assert is_deep_research_model(f"apodex/{model}") is False + + @pytest.mark.parametrize("model", DEEP_RESEARCH_MODELS) + def test_deep_research_models_are_detected(self, model: str): + assert is_deep_research_model(model) is True + assert is_deep_research_model(f"apodex/{model}") is True + + def test_prefix_does_not_leak_into_classification(self): + """A provider prefix containing the marker must not flip a core model.""" + assert is_deep_research_model("some-deep-gateway/apodex-1.1") is False + + +class TestCredentialResolution: + def test_defaults(self, monkeypatch: pytest.MonkeyPatch): + monkeypatch.delenv("APODEX_API_BASE", raising=False) + monkeypatch.setenv("APODEX_API_KEY", "sk-env") + + assert get_apodex_api_base(None) == APODEX_API_BASE_URL + assert get_apodex_api_key(None) == "sk-env" + + def test_explicit_values_win_over_env(self, monkeypatch: pytest.MonkeyPatch): + monkeypatch.setenv("APODEX_API_BASE", "https://env.apodex.test/v1") + monkeypatch.setenv("APODEX_API_KEY", "sk-env") + + assert get_apodex_api_base("https://explicit.apodex.test/v1") == "https://explicit.apodex.test/v1" + assert get_apodex_api_key("sk-explicit") == "sk-explicit" + + def test_env_base_overrides_default(self, monkeypatch: pytest.MonkeyPatch): + monkeypatch.setenv("APODEX_API_BASE", "https://env.apodex.test/v1") + assert get_apodex_api_base(None) == "https://env.apodex.test/v1" + + +class TestRegistration: + def test_provider_enum_and_lists(self): + assert LlmProviders.APODEX.value == "apodex" + assert "apodex" in litellm.provider_list + assert "apodex" in litellm.constants.openai_compatible_providers + assert APODEX_API_BASE_URL in litellm.constants.openai_compatible_endpoints + + def test_not_registered_as_a_json_provider(self): + """Apodex needs model-aware transformations, so it must not fall into the + generic JSON path, which would shadow the Python configs in provider resolution.""" + from litellm.llms.openai_like.json_loader import JSONProviderRegistry + + assert JSONProviderRegistry.exists("apodex") is False + + def test_config_classes_resolve_from_the_lazy_registry(self): + assert litellm.ApodexChatConfig().custom_llm_provider == "apodex" + assert litellm.ApodexResponsesConfig().custom_llm_provider == LlmProviders.APODEX + + def test_packaged_endpoint_matrix_matches_the_source(self): + source = json.loads((REPO_ROOT / "provider_endpoints_support.json").read_text()) + backup = json.loads((REPO_ROOT / "litellm" / "provider_endpoints_support_backup.json").read_text()) + + assert backup["providers"]["apodex"] == source["providers"]["apodex"] + + +class TestModelMetadata: + @pytest.fixture(scope="class") + def model_cost(self) -> dict: + with open(REPO_ROOT / "model_prices_and_context_window.json") as f: + return json.load(f) + + def test_every_apodex_model_is_registered(self, model_cost: dict): + assert {key for key in model_cost if key.startswith("apodex/")} == { + f"apodex/{model}" for model in (*CORE_MODELS, *DEEP_RESEARCH_MODELS) + } + + def test_core_model_pricing(self, model_cost: dict): + info = model_cost["apodex/apodex-1.1"] + assert info["litellm_provider"] == "apodex" + assert info["mode"] == "chat" + assert info["max_input_tokens"] == 262144 + # GET /v1/models reports max_completion_tokens 65536, well under the context window + assert info["max_output_tokens"] == 65536 + assert info["max_tokens"] == info["max_output_tokens"] + assert info["input_cost_per_token"] == 3e-07 + assert info["cache_read_input_token_cost"] == 3e-08 + assert info["output_cost_per_token"] == 3e-06 + # Requests over 200K input tokens are billed at 2x across every tier + assert info["input_cost_per_token_above_200k_tokens"] == 6e-07 + assert info["cache_read_input_token_cost_above_200k_tokens"] == 6e-08 + assert info["output_cost_per_token_above_200k_tokens"] == 6e-06 + assert info["supports_prompt_caching"] is True + assert info["supports_function_calling"] is True + assert info["supported_endpoints"] == ["/v1/chat/completions", "/v1/responses", "/v1/messages"] + + def test_deep_research_model_pricing(self, model_cost: dict): + info = model_cost["apodex/apodex-1-1-deep-research"] + assert info["max_input_tokens"] == 131072 + assert info["max_output_tokens"] == 65536 + assert info["input_cost_per_token"] == 5e-06 + assert info["output_cost_per_token"] == 2e-05 + assert info["supports_function_calling"] is False + assert info["supports_response_schema"] is False + assert info["supports_prompt_caching"] is False + assert info["supports_web_search"] is True + + @pytest.mark.parametrize("model", DEEP_RESEARCH_MODELS) + def test_deep_research_models_are_not_on_the_native_messages_path(self, model_cost: dict, model: str): + """Apodex serves /v1/messages for the core models only.""" + assert "/v1/messages" not in model_cost[f"apodex/{model}"]["supported_endpoints"] + + def test_discover_is_responses_only(self, model_cost: dict): + """The Discover tiers answer 400 unsupported_api on /v1/chat/completions.""" + assert model_cost["apodex/apodex-1-1-deep-discover"]["supported_endpoints"] == ["/v1/responses"] + + @pytest.mark.parametrize("model", ("apodex-1-1-deep-research", "apodex-1-1-deep-solve")) + def test_the_other_deep_tiers_keep_chat_completions(self, model_cost: dict, model: str): + assert model_cost[f"apodex/{model}"]["supported_endpoints"] == [ + "/v1/chat/completions", + "/v1/responses", + ] + + def test_backup_cost_map_in_sync(self, model_cost: dict): + with open(REPO_ROOT / "litellm" / "model_prices_and_context_window_backup.json") as f: + backup = json.load(f) + for key in (key for key in model_cost if key.startswith("apodex/")): + assert backup[key] == model_cost[key], f"{key} differs between main and backup cost maps" diff --git a/tests/test_litellm/llms/apodex/test_apodex_messages_transformation.py b/tests/test_litellm/llms/apodex/test_apodex_messages_transformation.py new file mode 100644 index 00000000000..ed6da35bba6 --- /dev/null +++ b/tests/test_litellm/llms/apodex/test_apodex_messages_transformation.py @@ -0,0 +1,128 @@ +""" +Apodex Anthropic Messages transformation. + +Apodex implements the Anthropic protocol natively at POST /v1/messages, but only +serves the core models there. The Deep Research tiers must keep working on the +same route through LiteLLM's translation instead of being handed to a path that +would reject them. +""" + +import pytest + +import litellm +from litellm.types.utils import LlmProviders +from litellm.utils import ProviderConfigManager + +CORE_MODELS = ("apodex-1.1", "apodex-1.1-mini") +DEEP_RESEARCH_MODELS = ( + "apodex-1-1-deep-research", + "apodex-1-1-deep-solve", + "apodex-1-1-deep-discover", +) + + +@pytest.fixture(autouse=True) +def _apodex_env(monkeypatch: pytest.MonkeyPatch): + monkeypatch.setenv("APODEX_API_KEY", "sk-apodex-test") + monkeypatch.delenv("APODEX_API_BASE", raising=False) + yield + + +def _messages_config(model: str): + return ProviderConfigManager.get_provider_anthropic_messages_config(model=model, provider=LlmProviders.APODEX) + + +def _complete_url() -> str: + """Resolve the endpoint the way the handler does: validate first, then build the URL. + + validate_anthropic_messages_environment returns the api_base the handler feeds + into get_complete_url, so the two steps have to run in that order. + """ + config = _messages_config("apodex-1.1") + assert config is not None + _, api_base = config.validate_anthropic_messages_environment( + headers={}, model="apodex-1.1", messages=[], optional_params={}, litellm_params={} + ) + return config.get_complete_url( + api_base=api_base, api_key=None, model="apodex-1.1", optional_params={}, litellm_params={} + ) + + +class TestNativePassthroughRouting: + @pytest.mark.parametrize("model", CORE_MODELS) + def test_core_models_get_the_native_config(self, model: str): + config = _messages_config(model) + assert config is not None + assert type(config).__name__ == "ApodexAnthropicMessagesConfig" + assert config.custom_llm_provider == "apodex" + assert config.should_strip_billing_metadata() is True + + @pytest.mark.parametrize("model", DEEP_RESEARCH_MODELS) + def test_deep_research_models_fall_back_to_translation(self, model: str): + """No native config means LiteLLM uses a protocol translation instead of + forwarding to a path Apodex does not serve for these tiers.""" + assert _messages_config(model) is None + + @pytest.mark.parametrize("stream", (False, True), ids=("non-streaming", "streaming")) + def test_deep_research_translation_uses_responses_api(self, monkeypatch: pytest.MonkeyPatch, stream: bool): + from litellm.llms.anthropic.experimental_pass_through.messages import handler + + captured: dict = {} + + class ResponsesRouteSelected(Exception): + pass + + def capture_responses_translation(**kwargs): + captured.update(kwargs) + raise ResponsesRouteSelected + + def reject_chat_translation(**kwargs): + pytest.fail("Apodex Deep Research messages must not route through chat completions") + + monkeypatch.setattr(litellm, "responses", capture_responses_translation) + monkeypatch.setattr(litellm, "completion", reject_chat_translation) + + with pytest.raises(ResponsesRouteSelected): + handler.anthropic_messages_handler( + max_tokens=256, + messages=[{"role": "user", "content": "hi"}], + model="apodex/apodex-1-1-deep-research", + custom_llm_provider="apodex", + stream=stream, + ) + + assert captured["model"] == "apodex-1-1-deep-research" + assert captured["custom_llm_provider"] == "apodex" + assert captured.get("stream", False) is stream + + +class TestNativePassthroughRequest: + def test_url_targets_the_native_messages_path(self): + assert _complete_url() == "https://api.apodex.ai/v1/messages" + + def test_url_honours_an_api_base_override(self, monkeypatch: pytest.MonkeyPatch): + monkeypatch.setenv("APODEX_API_BASE", "https://env.apodex.test/v1") + assert _complete_url() == "https://env.apodex.test/v1/messages" + + def test_headers_use_the_provider_api_key(self): + config = _messages_config("apodex-1.1") + assert config is not None + headers, _ = config.validate_anthropic_messages_environment( + headers={}, model="apodex-1.1", messages=[], optional_params={}, litellm_params={} + ) + assert headers["authorization"] == "Bearer sk-apodex-test" + assert headers["anthropic-version"] == "2023-06-01" + assert headers["content-type"] == "application/json" + + def test_caller_supplied_auth_header_is_not_overwritten(self): + config = _messages_config("apodex-1.1") + assert config is not None + headers, _ = config.validate_anthropic_messages_environment( + headers={"x-api-key": "sk-caller"}, + model="apodex-1.1", + messages=[], + optional_params={}, + litellm_params={}, + ) + assert headers["x-api-key"] == "sk-caller" + assert "authorization" not in headers diff --git a/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py b/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py new file mode 100644 index 00000000000..252a07bf607 --- /dev/null +++ b/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py @@ -0,0 +1,426 @@ +""" +Apodex Responses API transformation. + +The core models expose a stateless subset of /v1/responses while the Deep +Research tiers keep server-side state, so the parameter contract is keyed off +the model rather than applied provider-wide. +""" + +import gzip +from types import SimpleNamespace + +import httpx +import pytest + +import litellm +from litellm.llms.anthropic.experimental_pass_through.responses_adapters.streaming_iterator import ( + AnthropicResponsesStreamWrapper, +) +from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler +from litellm.types.utils import LlmProviders +from litellm.utils import ProviderConfigManager + +CORE_MODEL = "apodex/apodex-1.1" +CORE_MINI_MODEL = "apodex/apodex-1.1-mini" +DEEP_RESEARCH_MODEL = "apodex/apodex-1-1-deep-research" + + +@pytest.fixture(autouse=True) +def _apodex_env(monkeypatch: pytest.MonkeyPatch): + monkeypatch.setenv("APODEX_API_KEY", "sk-apodex-test") + monkeypatch.delenv("APODEX_API_BASE", raising=False) + monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") + monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) + monkeypatch.setattr(litellm, "drop_params", False) + yield + + +_SENTINEL = "apodex-request-captured" + + +def _capture(**kwargs) -> dict: + """Run litellm.responses() and return the request it would have sent. + + Validation errors raised before the request is built propagate to the caller. + """ + captured: dict = {} + + class CapturingHandler(HTTPHandler): + def post(self, *args, **post_kwargs): + captured.update(url=post_kwargs.get("url"), body=post_kwargs.get("json")) + raise RuntimeError(_SENTINEL) + + try: + litellm.responses(client=CapturingHandler(), **kwargs) + except Exception as exc: + if _SENTINEL not in str(exc): + raise + assert captured, "no request was sent" + return captured + + +def _responses_config(model: str): + return ProviderConfigManager.get_provider_responses_api_config(model=model, provider=LlmProviders.APODEX) + + +class TestConfigSelection: + def test_python_config_is_used_for_every_apodex_model(self): + for model in ("apodex-1.1", "apodex-1.1-mini", "apodex-1-1-deep-research"): + config = _responses_config(model) + assert type(config).__name__ == "ApodexResponsesConfig" + + def test_auth_uses_the_apodex_key(self): + config = _responses_config("apodex-1.1") + assert config.validate_environment(headers={}, model="apodex-1.1", litellm_params=None) == { + "Content-Type": "application/json", + "Authorization": "Bearer sk-apodex-test", + } + + def test_auth_does_not_fall_back_to_an_openai_key(self, monkeypatch: pytest.MonkeyPatch): + """The inherited OpenAI config would forward OPENAI_API_KEY to Apodex.""" + monkeypatch.delenv("APODEX_API_KEY", raising=False) + monkeypatch.setenv("OPENAI_API_KEY", "sk-openai-must-not-leak") + + config = _responses_config("apodex-1.1") + assert config.validate_environment(headers={}, model="apodex-1.1", litellm_params=None) == {} + + def test_request_targets_the_apodex_responses_url(self): + assert _capture(model=CORE_MODEL, input="hi")["url"] == "https://api.apodex.ai/v1/responses" + + def test_polling_without_model_resolution_targets_apodex(self): + config = _responses_config("apodex-1-1-deep-research") + assert config.get_complete_url(api_base=None, litellm_params={}) == "https://api.apodex.ai/v1/responses" + + def test_polling_honours_an_explicit_api_base(self): + config = _responses_config("apodex-1-1-deep-research") + assert ( + config.get_complete_url(api_base="https://gateway.apodex.test/v1/", litellm_params={}) + == "https://gateway.apodex.test/v1/responses" + ) + + def test_request_honours_an_api_base_override(self, monkeypatch: pytest.MonkeyPatch): + monkeypatch.setenv("APODEX_API_BASE", "https://env.apodex.test/v1") + assert _capture(model=CORE_MODEL, input="hi")["url"] == "https://env.apodex.test/v1/responses" + + +class TestStreamDefault: + def test_non_streaming_pins_stream_false(self): + captured = _capture(model=DEEP_RESEARCH_MODEL, input="hi") + assert captured["url"] == "https://api.apodex.ai/v1/responses" + assert captured["body"]["stream"] is False + + @pytest.mark.asyncio + async def test_streaming_sends_stream_true(self): + captured: dict = {} + + class CapturingHandler(AsyncHTTPHandler): + async def post(self, *args, **kwargs): + captured.update(body=kwargs.get("json")) + raise RuntimeError("captured") + + with pytest.raises(Exception, match="captured"): + await litellm.aresponses(model=DEEP_RESEARCH_MODEL, input="hi", stream=True, client=CapturingHandler()) + + assert captured["body"]["stream"] is True + + +class TestCoreModelStatelessSubset: + """Apodex core models reject anything that would persist state on their side.""" + + @pytest.mark.parametrize("model", [CORE_MODEL, CORE_MINI_MODEL]) + def test_store_is_pinned_false(self, model: str): + captured = _capture(model=model, input="hi") + assert captured["body"]["store"] is False + + def test_store_true_raises(self): + with pytest.raises(litellm.UnsupportedParamsError, match="store=True"): + _capture(model=CORE_MODEL, input="hi", store=True) + + def test_background_raises(self): + with pytest.raises(litellm.UnsupportedParamsError, match="background"): + _capture(model=CORE_MODEL, input="hi", background=True) + + def test_previous_response_id_raises(self): + with pytest.raises(litellm.UnsupportedParamsError, match="previous_response_id"): + _capture(model=CORE_MODEL, input="hi", previous_response_id="resp_1") + + def test_drop_params_strips_all_three(self, monkeypatch: pytest.MonkeyPatch): + monkeypatch.setattr(litellm, "drop_params", True) + captured = _capture( + model=CORE_MODEL, + input="hi", + store=True, + background=True, + previous_response_id="resp_1", + ) + + assert captured["body"]["store"] is False + assert "background" not in captured["body"] + assert "previous_response_id" not in captured["body"] + + def test_stateful_params_are_not_advertised(self): + supported = _responses_config("apodex-1.1").get_supported_openai_params("apodex-1.1") + assert "background" not in supported + assert "previous_response_id" not in supported + assert "max_output_tokens" in supported + + def test_max_output_tokens_still_passes_through(self): + captured = _capture(model=CORE_MODEL, input="hi", max_output_tokens=512) + assert captured["body"]["max_output_tokens"] == 512 + + +class TestDeepResearchKeepsState: + """The agent tiers survive client disconnects, so none of this may be stripped.""" + + def test_background_passes_through(self): + captured = _capture(model=DEEP_RESEARCH_MODEL, input="hi", background=True) + assert captured["body"]["background"] is True + + def test_store_and_previous_response_id_pass_through(self): + captured = _capture(model=DEEP_RESEARCH_MODEL, input="hi", store=True, previous_response_id="resp_1") + assert captured["body"]["store"] is True + assert captured["body"]["previous_response_id"] == "resp_1" + + def test_store_is_not_pinned_when_unset(self): + captured = _capture(model=DEEP_RESEARCH_MODEL, input="hi") + assert "store" not in captured["body"] + + def test_stateful_params_are_advertised(self): + supported = _responses_config("apodex-1-1-deep-research").get_supported_openai_params( + "apodex-1-1-deep-research" + ) + assert "background" in supported + assert "previous_response_id" in supported + + def test_minimal_cancel_response_is_normalized(self): + config = _responses_config("apodex-1-1-deep-research") + response = config.transform_cancel_response_api_response( + raw_response=httpx.Response( + 200, + json={"id": "resp_1", "object": "response", "status": "cancelled"}, + ), + logging_obj=None, + ) + + assert response.id == "resp_1" + assert response.status == "cancelled" + assert response.output == [] + assert response.created_at > 0 + + def test_cancel_response_survives_a_compressed_upstream_response(self): + """The body is rebuilt, so the original framing headers must not follow it. + + httpx decodes on read, so carrying Content-Encoding over from the compressed + upstream response makes it try to gunzip the plain JSON replacement. + """ + body = b'{"id": "resp_1", "object": "response", "status": "cancelled"}' + # As it arrives off the wire: httpx decodes the body but leaves the header in place + upstream = httpx.Response( + 200, + headers={ + "content-encoding": "gzip", + "x-ratelimit-remaining-requests": "42", + }, + content=gzip.compress(body), + ) + assert upstream.content == body + + config = _responses_config("apodex-1-1-deep-research") + response = config.transform_cancel_response_api_response( + raw_response=upstream, + logging_obj=None, + ) + + assert response.id == "resp_1" + assert response.status == "cancelled" + assert response._hidden_params["headers"]["x-ratelimit-remaining-requests"] == "42" + + def test_output_text_delta_is_normalized(self): + """A Deep Research stream never emits response.output_text.delta of its own. + + Payload shape captured from a live stream; the extra swarm.data keys ride + along untouched and must not affect the mapping. + """ + config = _responses_config("apodex-1-1-deep-research") + event = config.transform_streaming_response( + model="apodex-1-1-deep-research", + parsed_chunk={ + "type": "response.swarm.llm_delta", + "created_at": 1786873219.3543231, + "response_id": "w_c4b77c96", + "sequence_number": 12, + "swarm": { + "agent_id": "reporter", + "data": { + "channel": "output_text", + "delta": "Hello there, friend!", + "delta_index": 0, + "call_id": "llm_fce8e965", + "turn": 1, + }, + }, + }, + logging_obj=None, + ) + + assert event.type == "response.output_text.delta" + assert event.item_id == "msg_w_c4b77c96" + assert event.delta == "Hello there, friend!" + assert event.sequence_number == 12 + assert event.content_index == 0 + + def test_reasoning_delta_becomes_a_reasoning_summary_delta(self): + """`response.reasoning_summary_text.delta` is what LiteLLM already translates + into an Anthropic `thinking_delta`, which is the route Deep Research takes on + /v1/messages. It also keeps a separate item id from the answer text.""" + config = _responses_config("apodex-1-1-deep-research") + event = config.transform_streaming_response( + model="apodex-1-1-deep-research", + parsed_chunk={ + "type": "response.swarm.llm_delta", + "response_id": "w_c4b77c96", + "sequence_number": 7, + "swarm": { + "agent_id": "stateful_react", + "data": {"channel": "reasoning", "delta": "The user wants", "delta_index": 0}, + }, + }, + logging_obj=None, + ) + + assert event.type == "response.reasoning_summary_text.delta" + assert event.item_id == "rs_w_c4b77c96" + assert event.delta == "The user wants" + assert event.summary_index == 0 + assert not hasattr(event, "content_index") + + def test_reasoning_and_answer_form_valid_anthropic_blocks(self): + config = _responses_config("apodex-1-1-deep-research") + reasoning_event = config.transform_streaming_response( + model="apodex-1-1-deep-research", + parsed_chunk={ + "type": "response.swarm.llm_delta", + "response_id": "w_c4b77c96", + "sequence_number": 7, + "swarm": {"data": {"channel": "reasoning", "delta": "The user wants"}}, + }, + logging_obj=None, + ) + answer_event = config.transform_streaming_response( + model="apodex-1-1-deep-research", + parsed_chunk={ + "type": "response.swarm.llm_delta", + "response_id": "w_c4b77c96", + "sequence_number": 8, + "swarm": {"data": {"channel": "output_text", "delta": "Hello there, friend!"}}, + }, + logging_obj=None, + ) + wrapper = AnthropicResponsesStreamWrapper(responses_stream=None, model="apodex-1-1-deep-research") + for event in ( + reasoning_event, + answer_event, + {"type": "response.completed", "response": SimpleNamespace(status="completed", output=[], usage=None)}, + ): + wrapper._process_event(event) + + chunks = list(wrapper._chunk_queue) + assert [(chunk["type"], chunk.get("index")) for chunk in chunks] == [ + ("content_block_start", 0), + ("content_block_delta", 0), + ("content_block_stop", 0), + ("content_block_start", 1), + ("content_block_delta", 1), + ("content_block_stop", 1), + ("message_delta", None), + ("message_stop", None), + ] + assert chunks[0]["content_block"]["type"] == "thinking" + assert chunks[1]["delta"] == {"type": "thinking_delta", "thinking": "The user wants"} + assert chunks[3]["content_block"]["type"] == "text" + assert chunks[4]["delta"] == {"type": "text_delta", "text": "Hello there, friend!"} + + @pytest.mark.parametrize( + "channel", + (None, "tool_output", []), + ids=("no-channel", "unknown-channel", "non-string-channel"), + ) + def test_intermediate_agent_deltas_are_not_claimed(self, channel): + """The worker agent streams a draft answer on a channel-less delta. + + Live capture: those four deltas spell "Hello, friend! How are you?" while the + reporter's `output_text` is the "Hello there, friend!" that lands in + response.completed. Claiming them would splice the draft into the answer. + """ + data = {"delta": "Hello,"} if channel is None else {"channel": channel, "delta": "Hello,"} + config = _responses_config("apodex-1-1-deep-research") + event = config.transform_streaming_response( + model="apodex-1-1-deep-research", + parsed_chunk={ + "type": "response.swarm.llm_delta", + "response_id": "w_c4b77c96", + "sequence_number": 7, + "swarm": {"agent_id": "stateful_react", "data": data}, + }, + logging_obj=None, + ) + + assert event.type == "response.swarm.llm_delta" + + def test_non_string_delta_is_not_claimed(self): + config = _responses_config("apodex-1-1-deep-research") + assert ( + config._map_swarm_delta( + { + "type": "response.swarm.llm_delta", + "swarm": {"data": {"channel": "output_text", "delta": None}}, + } + ) + is None + ) + + @pytest.mark.parametrize( + "event_type", + ( + "response.swarm.run_started", + "response.swarm.run_finished", + "response.swarm.injection_window", + "response.swarm.llm_attempt_started", + "response.swarm.llm_attempt_finished", + ), + ) + def test_swarm_lifecycle_events_pass_through(self, event_type: str): + """LiteLLM cannot drop a chunk from the stream, so these stay as GenericEvent + rather than being silently swallowed; `run_finished` carries the final content.""" + config = _responses_config("apodex-1-1-deep-research") + event = config.transform_streaming_response( + model="apodex-1-1-deep-research", + parsed_chunk={ + "type": event_type, + "response_id": "w_c4b77c96", + "sequence_number": 4, + "swarm": {"agent_id": "reporter", "data": {"status": "success"}}, + }, + logging_obj=None, + ) + + assert event.type == event_type + + def test_documented_events_pass_through_untouched(self): + config = _responses_config("apodex-1-1-deep-research") + event = config.transform_streaming_response( + model="apodex-1-1-deep-research", + parsed_chunk={"type": "response.in_progress", "sequence_number": 2}, + logging_obj=None, + ) + + assert event.type == "response.in_progress" + + def test_non_json_cancel_body_raises_the_provider_error(self): + """The gateway answers a timed-out cancel with an HTML 504, not the JSON envelope.""" + config = _responses_config("apodex-1-1-deep-research") + with pytest.raises(Exception, match="gateway timeout"): + config.transform_cancel_response_api_response( + raw_response=httpx.Response(504, content=b"gateway timeout"), + logging_obj=None, + )