From 20ef10e920d3f152e863523c5a8ea9e05f776a34 Mon Sep 17 00:00:00 2001 From: zhanghanduo Date: Sun, 16 Aug 2026 11:19:57 +0800 Subject: [PATCH 01/12] feat(providers): add Apodex as an OpenAI-compatible provider Registers apodex via providers.json with /v1/chat/completions, /v1/responses and native /v1/messages, plus price map entries for the two core models and the six deep research tiers. Apodex defaults `stream` to true on both /v1/chat/completions and /v1/responses, so a non-streaming litellm call would get SSE back and fail to parse it. Adds a `send_explicit_stream_false` special-handling flag that pins the field on the wire, and rewrites the JSON provider param mapping to build its result instead of mutating the caller's dict. --- README.md | 1 + basedpyright-code-budget.json | 4 +- litellm/constants.py | 2 + .../get_llm_provider_logic.py | 3 + litellm/llms/openai_like/README.md | 10 +- litellm/llms/openai_like/dynamic_config.py | 72 ++-- litellm/llms/openai_like/providers.json | 12 + ...odel_prices_and_context_window_backup.json | 190 +++++++++++ litellm/types/utils.py | 1 + model_prices_and_context_window.json | 190 +++++++++++ provider_endpoints_support.json | 17 + .../llms/openai_like/test_apodex_provider.py | 310 ++++++++++++++++++ .../llms/openai_like/test_json_providers.py | 57 ++++ type-discipline-budget.json | 4 +- 14 files changed, 838 insertions(+), 35 deletions(-) create mode 100644 tests/test_litellm/llms/openai_like/test_apodex_provider.py diff --git a/README.md b/README.md index 32b0160dbaa..265038aaa85 100644 --- a/README.md +++ b/README.md @@ -279,6 +279,7 @@ curl -X POST 'http://0.0.0.0:4000/v1/chat/completions' \ | [Anthropic (`anthropic`)](https://docs.litellm.ai/docs/providers/anthropic) | ✅ | ✅ | ✅ | | | | | | ✅ | | | [Anthropic Text (`anthropic_text`)](https://docs.litellm.ai/docs/providers/anthropic) | ✅ | ✅ | ✅ | | | | | | ✅ | | | [Anyscale](https://docs.litellm.ai/docs/providers/anyscale) | ✅ | ✅ | ✅ | | | | | | | | +| [Apodex (`apodex`)](https://docs.litellm.ai/docs/providers/apodex) | ✅ | ✅ | ✅ | | | | | | | | | [AssemblyAI (`assemblyai`)](https://docs.litellm.ai/docs/pass_through/assembly_ai) | ✅ | ✅ | ✅ | | | ✅ | | | | | | [Auto Router (`auto_router`)](https://docs.litellm.ai/docs/proxy/auto_routing) | ✅ | ✅ | ✅ | | | | | | | | | [AWS - Bedrock (`bedrock`)](https://docs.litellm.ai/docs/providers/bedrock) | ✅ | ✅ | ✅ | ✅ | | | | | | ✅ | diff --git a/basedpyright-code-budget.json b/basedpyright-code-budget.json index 06010c706e3..0109927f2c6 100644 --- a/basedpyright-code-budget.json +++ b/basedpyright-code-budget.json @@ -99,7 +99,7 @@ "limit": 0 }, "reportUnknownArgumentType": { - "limit": 44776 + "limit": 44774 }, "reportUnknownLambdaType": { "limit": 113 @@ -111,7 +111,7 @@ "limit": 19967 }, "reportUnknownVariableType": { - "limit": 30881 + "limit": 30879 }, "reportUnnecessaryCast": { "limit": 117 diff --git a/litellm/constants.py b/litellm/constants.py index 8f236eba327..7c99b0c9c46 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -756,6 +756,7 @@ openai_compatible_endpoints: Final[list] = [ "https://api.libertai.io/v1", "https://pinstripes.io/v1", "https://api.meta.ai/v1", + "https://api.apodex.ai/v1", ] @@ -823,6 +824,7 @@ openai_compatible_providers: Final[list] = [ "pinstripes", # Pinstripes - JSON-configured provider "darkbloom", "meta", # Meta Model API (Muse Spark) - JSON-configured provider + "apodex", # Apodex - JSON-configured provider ] openai_text_completion_compatible_providers: Final[list] = [ # providers that support `/v1/completions` "together_ai", diff --git a/litellm/litellm_core_utils/get_llm_provider_logic.py b/litellm/litellm_core_utils/get_llm_provider_logic.py index dbb40913e14..9f07710007f 100644 --- a/litellm/litellm_core_utils/get_llm_provider_logic.py +++ b/litellm/litellm_core_utils/get_llm_provider_logic.py @@ -349,6 +349,9 @@ def get_llm_provider( elif endpoint == "https://api.meta.ai/v1": custom_llm_provider = "meta" dynamic_api_key = get_secret_str("META_API_KEY") + elif endpoint == "https://api.apodex.ai/v1": + custom_llm_provider = "apodex" + dynamic_api_key = get_secret_str("APODEX_API_KEY") if api_base is not None and not isinstance(api_base, str): raise Exception(f"api base needs to be a string. api_base={api_base}") diff --git a/litellm/llms/openai_like/README.md b/litellm/llms/openai_like/README.md index e9aaafe48a1..fc9375f2efb 100644 --- a/litellm/llms/openai_like/README.md +++ b/litellm/llms/openai_like/README.md @@ -59,7 +59,15 @@ That's it! The provider will be automatically loaded and available. // Optional: Special handling flags "special_handling": { - "convert_content_list_to_string": true + "convert_content_list_to_string": true, + + // Send "stream": false explicitly instead of omitting it. Needed by + // providers whose /v1/chat/completions and /v1/responses default to + // streaming, where omitting the field returns SSE to a non-streaming call + "send_explicit_stream_false": true, + + // Always send "store": false on /v1/responses + "force_store_false": true } } } diff --git a/litellm/llms/openai_like/dynamic_config.py b/litellm/llms/openai_like/dynamic_config.py index 19e29bcdcb2..52168e8bb66 100644 --- a/litellm/llms/openai_like/dynamic_config.py +++ b/litellm/llms/openai_like/dynamic_config.py @@ -2,7 +2,7 @@ Dynamic configuration class generator for JSON-based providers. """ -from collections.abc import Coroutine +from collections.abc import Coroutine, Mapping from typing import Any, Final, Literal, overload from litellm._logging import verbose_logger @@ -17,6 +17,17 @@ from litellm.types.llms.openai import AllMessageValues from .json_loader import SimpleProviderConfig +def _clamp_temperature(temperature: float, n: int, constraints: Mapping[str, float]) -> float: + capped: Final = ( + min(temperature, constraints["temperature_max"]) if "temperature_max" in constraints else temperature + ) + floored: Final = max(capped, constraints["temperature_min"]) if "temperature_min" in constraints else capped + floor_for_multiple_choices: Final = constraints.get("temperature_min_with_n_gt_1") + if n > 1 and floor_for_multiple_choices is not None: + return max(floored, floor_for_multiple_choices) + return floored + + def create_config_class(provider: SimpleProviderConfig): """Generate config class dynamically from JSON configuration""" @@ -131,37 +142,36 @@ def create_config_class(provider: SimpleProviderConfig): """Apply parameter mappings and constraints""" supported_params: Final = self.get_supported_openai_params(model) + mapped: Final = { + **optional_params, + **{ + provider.param_mappings.get(param, param): value + for param, value in non_default_params.items() + if param in provider.param_mappings or param in supported_params + }, + } - # Apply supported params - for param, value in non_default_params.items(): - # Check parameter mappings first - if param in provider.param_mappings: - optional_params[provider.param_mappings[param]] = value - elif param in supported_params: - optional_params[param] = value + constrained: Final = ( + mapped + if "temperature" not in mapped + else { + **mapped, + "temperature": _clamp_temperature( + temperature=mapped["temperature"], + n=mapped.get("n", 1), + constraints=provider.constraints, + ), + } + ) - # Apply temperature constraints if present - if "temperature" in optional_params: - temp = optional_params["temperature"] - constraints: Final = provider.constraints - - # Clamp to max - if "temperature_max" in constraints: - temp = min(temp, constraints["temperature_max"]) - - # Clamp to min - if "temperature_min" in constraints: - temp = max(temp, constraints["temperature_min"]) - - # Special case: temperature_min_with_n_gt_1 - if "temperature_min_with_n_gt_1" in constraints: - n: Final = optional_params.get("n", 1) - if n > 1 and temp < constraints["temperature_min_with_n_gt_1"]: - temp = constraints["temperature_min_with_n_gt_1"] - - optional_params["temperature"] = temp - - return optional_params + # The OpenAI SDK omits `stream` entirely when it is false, which makes + # stream-by-default providers answer a non-streaming call with SSE. Pin it + # on the wire through extra_body, which the SDK merges into the request body. + if not provider.special_handling.get("send_explicit_stream_false") or constrained.get("stream"): + return constrained + requested_extra_body: Final = constrained.get("extra_body") + extra_body: Final[dict] = requested_extra_body if isinstance(requested_extra_body, dict) else {} + return {**constrained, "extra_body": {"stream": False, **extra_body}} @property def custom_llm_provider(self) -> str | None: @@ -232,6 +242,8 @@ def create_responses_config_class(provider: SimpleProviderConfig): ) -> dict: if provider.special_handling.get("force_store_false"): response_api_optional_request_params["store"] = False + if provider.special_handling.get("send_explicit_stream_false"): + response_api_optional_request_params.setdefault("stream", False) return super().transform_responses_api_request( model=model, input=input, diff --git a/litellm/llms/openai_like/providers.json b/litellm/llms/openai_like/providers.json index 164100d4194..18cf49e6ccc 100644 --- a/litellm/llms/openai_like/providers.json +++ b/litellm/llms/openai_like/providers.json @@ -175,6 +175,18 @@ "base_class": "openai_gpt", "supported_endpoints": ["/v1/chat/completions", "/v1/responses", "/v1/messages"] }, + "apodex": { + "base_url": "https://api.apodex.ai/v1", + "api_key_env": "APODEX_API_KEY", + "api_base_env": "APODEX_API_BASE", + "param_mappings": { + "max_completion_tokens": "max_tokens" + }, + "supported_endpoints": ["/v1/chat/completions", "/v1/responses", "/v1/messages"], + "special_handling": { + "send_explicit_stream_false": true + } + }, "pinstripes": { "base_url": "https://pinstripes.io/v1", "api_key_env": "PINSTRIPES_API_KEY", diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index e6c6cab0631..9d6a66e9327 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -48296,6 +48296,196 @@ ], "supports_audio_output": true }, + "apodex/apodex-1.1": { + "max_tokens": 262144, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "input_cost_per_token": 3e-07, + "cache_read_input_token_cost": 3e-08, + "output_cost_per_token": 3e-06, + "input_cost_per_token_above_200k_tokens": 6e-07, + "cache_read_input_token_cost_above_200k_tokens": 6e-08, + "output_cost_per_token_above_200k_tokens": 6e-06, + "litellm_provider": "apodex", + "mode": "chat", + "source": "https://platform.apodex.ai/docs/models", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_vision": false + }, + "apodex/apodex-1.1-mini": { + "max_tokens": 262144, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "input_cost_per_token": 1e-07, + "cache_read_input_token_cost": 1e-08, + "output_cost_per_token": 1e-06, + "input_cost_per_token_above_200k_tokens": 2e-07, + "cache_read_input_token_cost_above_200k_tokens": 2e-08, + "output_cost_per_token_above_200k_tokens": 2e-06, + "litellm_provider": "apodex", + "mode": "chat", + "source": "https://platform.apodex.ai/docs/models", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_vision": false + }, + "apodex/apodex-1-1-deep-research": { + "max_tokens": 65536, + "max_input_tokens": 131072, + "max_output_tokens": 65536, + "input_cost_per_token": 5e-06, + "output_cost_per_token": 2e-05, + "litellm_provider": "apodex", + "mode": "chat", + "source": "https://platform.apodex.ai/docs/pricing", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supports_function_calling": false, + "supports_native_streaming": true, + "supports_prompt_caching": false, + "supports_reasoning": true, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_tool_choice": false, + "supports_vision": false, + "supports_web_search": true + }, + "apodex/apodex-1-1-deep-solve": { + "max_tokens": 65536, + "max_input_tokens": 131072, + "max_output_tokens": 65536, + "input_cost_per_token": 5e-06, + "output_cost_per_token": 2.5e-05, + "litellm_provider": "apodex", + "mode": "chat", + "source": "https://platform.apodex.ai/docs/pricing", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supports_function_calling": false, + "supports_native_streaming": true, + "supports_prompt_caching": false, + "supports_reasoning": true, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_tool_choice": false, + "supports_vision": false, + "supports_web_search": true + }, + "apodex/apodex-1-1-deep-discover": { + "max_tokens": 262144, + "max_input_tokens": 131072, + "max_output_tokens": 262144, + "input_cost_per_token": 1e-05, + "output_cost_per_token": 0.0001, + "litellm_provider": "apodex", + "mode": "chat", + "source": "https://platform.apodex.ai/docs/pricing", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supports_function_calling": false, + "supports_native_streaming": true, + "supports_prompt_caching": false, + "supports_reasoning": true, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_tool_choice": false, + "supports_vision": false, + "supports_web_search": true + }, + "apodex/apodex-1-0-deep-research": { + "max_tokens": 16384, + "max_input_tokens": 262144, + "max_output_tokens": 16384, + "input_cost_per_token": 1e-05, + "output_cost_per_token": 4e-05, + "litellm_provider": "apodex", + "mode": "chat", + "source": "https://platform.apodex.ai/docs/pricing", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supports_function_calling": false, + "supports_native_streaming": true, + "supports_prompt_caching": false, + "supports_reasoning": true, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_tool_choice": false, + "supports_vision": false, + "supports_web_search": true + }, + "apodex/apodex-1-0-deep-solve": { + "max_tokens": 16384, + "max_input_tokens": 262144, + "max_output_tokens": 16384, + "input_cost_per_token": 1e-05, + "output_cost_per_token": 5e-05, + "litellm_provider": "apodex", + "mode": "chat", + "source": "https://platform.apodex.ai/docs/pricing", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supports_function_calling": false, + "supports_native_streaming": true, + "supports_prompt_caching": false, + "supports_reasoning": true, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_tool_choice": false, + "supports_vision": false, + "supports_web_search": true + }, + "apodex/apodex-1-0-deep-discover": { + "max_tokens": 262144, + "max_input_tokens": 131072, + "max_output_tokens": 262144, + "input_cost_per_token": 1e-05, + "output_cost_per_token": 0.0001, + "litellm_provider": "apodex", + "mode": "chat", + "source": "https://platform.apodex.ai/docs/pricing", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supports_function_calling": false, + "supports_native_streaming": true, + "supports_prompt_caching": false, + "supports_reasoning": true, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_tool_choice": false, + "supports_vision": false, + "supports_web_search": true + }, "fallback_generalizations": { "rules": [ { diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 272fbabf807..61514db28fc 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -3727,6 +3727,7 @@ class LlmProviders(str, Enum): PINSTRIPES = "pinstripes" DARKBLOOM = "darkbloom" META = "meta" + APODEX = "apodex" LITELLM_AGENT = "litellm_agent" CURSOR = "cursor" BEDROCK_MANTLE = "bedrock_mantle" diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index e6c6cab0631..9d6a66e9327 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -48296,6 +48296,196 @@ ], "supports_audio_output": true }, + "apodex/apodex-1.1": { + "max_tokens": 262144, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "input_cost_per_token": 3e-07, + "cache_read_input_token_cost": 3e-08, + "output_cost_per_token": 3e-06, + "input_cost_per_token_above_200k_tokens": 6e-07, + "cache_read_input_token_cost_above_200k_tokens": 6e-08, + "output_cost_per_token_above_200k_tokens": 6e-06, + "litellm_provider": "apodex", + "mode": "chat", + "source": "https://platform.apodex.ai/docs/models", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_vision": false + }, + "apodex/apodex-1.1-mini": { + "max_tokens": 262144, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "input_cost_per_token": 1e-07, + "cache_read_input_token_cost": 1e-08, + "output_cost_per_token": 1e-06, + "input_cost_per_token_above_200k_tokens": 2e-07, + "cache_read_input_token_cost_above_200k_tokens": 2e-08, + "output_cost_per_token_above_200k_tokens": 2e-06, + "litellm_provider": "apodex", + "mode": "chat", + "source": "https://platform.apodex.ai/docs/models", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_vision": false + }, + "apodex/apodex-1-1-deep-research": { + "max_tokens": 65536, + "max_input_tokens": 131072, + "max_output_tokens": 65536, + "input_cost_per_token": 5e-06, + "output_cost_per_token": 2e-05, + "litellm_provider": "apodex", + "mode": "chat", + "source": "https://platform.apodex.ai/docs/pricing", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supports_function_calling": false, + "supports_native_streaming": true, + "supports_prompt_caching": false, + "supports_reasoning": true, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_tool_choice": false, + "supports_vision": false, + "supports_web_search": true + }, + "apodex/apodex-1-1-deep-solve": { + "max_tokens": 65536, + "max_input_tokens": 131072, + "max_output_tokens": 65536, + "input_cost_per_token": 5e-06, + "output_cost_per_token": 2.5e-05, + "litellm_provider": "apodex", + "mode": "chat", + "source": "https://platform.apodex.ai/docs/pricing", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supports_function_calling": false, + "supports_native_streaming": true, + "supports_prompt_caching": false, + "supports_reasoning": true, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_tool_choice": false, + "supports_vision": false, + "supports_web_search": true + }, + "apodex/apodex-1-1-deep-discover": { + "max_tokens": 262144, + "max_input_tokens": 131072, + "max_output_tokens": 262144, + "input_cost_per_token": 1e-05, + "output_cost_per_token": 0.0001, + "litellm_provider": "apodex", + "mode": "chat", + "source": "https://platform.apodex.ai/docs/pricing", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supports_function_calling": false, + "supports_native_streaming": true, + "supports_prompt_caching": false, + "supports_reasoning": true, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_tool_choice": false, + "supports_vision": false, + "supports_web_search": true + }, + "apodex/apodex-1-0-deep-research": { + "max_tokens": 16384, + "max_input_tokens": 262144, + "max_output_tokens": 16384, + "input_cost_per_token": 1e-05, + "output_cost_per_token": 4e-05, + "litellm_provider": "apodex", + "mode": "chat", + "source": "https://platform.apodex.ai/docs/pricing", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supports_function_calling": false, + "supports_native_streaming": true, + "supports_prompt_caching": false, + "supports_reasoning": true, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_tool_choice": false, + "supports_vision": false, + "supports_web_search": true + }, + "apodex/apodex-1-0-deep-solve": { + "max_tokens": 16384, + "max_input_tokens": 262144, + "max_output_tokens": 16384, + "input_cost_per_token": 1e-05, + "output_cost_per_token": 5e-05, + "litellm_provider": "apodex", + "mode": "chat", + "source": "https://platform.apodex.ai/docs/pricing", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supports_function_calling": false, + "supports_native_streaming": true, + "supports_prompt_caching": false, + "supports_reasoning": true, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_tool_choice": false, + "supports_vision": false, + "supports_web_search": true + }, + "apodex/apodex-1-0-deep-discover": { + "max_tokens": 262144, + "max_input_tokens": 131072, + "max_output_tokens": 262144, + "input_cost_per_token": 1e-05, + "output_cost_per_token": 0.0001, + "litellm_provider": "apodex", + "mode": "chat", + "source": "https://platform.apodex.ai/docs/pricing", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supports_function_calling": false, + "supports_native_streaming": true, + "supports_prompt_caching": false, + "supports_reasoning": true, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_tool_choice": false, + "supports_vision": false, + "supports_web_search": true + }, "fallback_generalizations": { "rules": [ { diff --git a/provider_endpoints_support.json b/provider_endpoints_support.json index 0712e8e383d..ce37e5dc243 100644 --- a/provider_endpoints_support.json +++ b/provider_endpoints_support.json @@ -177,6 +177,23 @@ "interactions": true } }, + "apodex": { + "display_name": "Apodex (`apodex`)", + "url": "https://docs.litellm.ai/docs/providers/apodex", + "endpoints": { + "chat_completions": true, + "messages": true, + "responses": true, + "embeddings": false, + "image_generations": false, + "audio_transcriptions": false, + "audio_speech": false, + "moderations": false, + "batches": false, + "rerank": false, + "a2a": false + } + }, "apertis": { "display_name": "Apertis (`apertis`)", "endpoints": { diff --git a/tests/test_litellm/llms/openai_like/test_apodex_provider.py b/tests/test_litellm/llms/openai_like/test_apodex_provider.py new file mode 100644 index 00000000000..cdcf9b4b50a --- /dev/null +++ b/tests/test_litellm/llms/openai_like/test_apodex_provider.py @@ -0,0 +1,310 @@ +""" +Tests for the Apodex provider (https://platform.apodex.ai/docs). +""" + +import json +from pathlib import Path + +import httpx +import openai +import pytest + +import litellm +from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler +from litellm.types.utils import LlmProviders +from litellm.utils import ProviderConfigManager + +REPO_ROOT = Path(__file__).parents[4] +CORE_MODEL = "apodex/apodex-1.1" +DEEP_RESEARCH_MODEL = "apodex/apodex-1-1-deep-research" + +CHAT_RESPONSE = { + "id": "chatcmpl-abc123", + "object": "chat.completion", + "created": 1712345678, + "model": "apodex-1.1", + "choices": [ + { + "index": 0, + "message": {"role": "assistant", "content": "ok", "reasoning_content": "let me think"}, + "finish_reason": "stop", + } + ], + "usage": { + "prompt_tokens": 1000, + "completion_tokens": 100, + "total_tokens": 1100, + "prompt_tokens_details": {"cached_tokens": 500}, + }, +} + +STREAM_BODY = ( + b'data: {"id":"chatcmpl-abc123","object":"chat.completion.chunk","created":1,"model":"apodex-1.1",' + b'"choices":[{"index":0,"delta":{"content":"ok"},"finish_reason":null}]}\n\n' + b"data: [DONE]\n\n" +) + + +@pytest.fixture(autouse=True) +def _apodex_env(monkeypatch: pytest.MonkeyPatch): + """Resolve models against the in-repo cost map, not the published one.""" + monkeypatch.setenv("APODEX_API_KEY", "sk-apodex-test") + monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") + monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) + yield + + +def _openai_client(captured: dict, *, stream: bool = False) -> openai.OpenAI: + def handler(request: httpx.Request) -> httpx.Response: + captured["url"] = str(request.url) + captured["body"] = json.loads(request.content) + if stream: + return httpx.Response(200, headers={"content-type": "text/event-stream"}, content=STREAM_BODY) + return httpx.Response(200, json=CHAT_RESPONSE) + + return openai.OpenAI( + api_key="sk-apodex-test", + base_url="https://api.apodex.ai/v1", + http_client=httpx.Client(transport=httpx.MockTransport(handler)), + ) + + +class TestApodexRegistration: + def test_provider_enum_and_lists(self): + assert LlmProviders.APODEX.value == "apodex" + assert "apodex" in litellm.provider_list + assert "apodex" in litellm.constants.openai_compatible_providers + + def test_json_provider_config(self): + from litellm.llms.openai_like.json_loader import JSONProviderRegistry + + apodex = JSONProviderRegistry.get("apodex") + assert apodex is not None + assert apodex.base_url == "https://api.apodex.ai/v1" + assert apodex.api_key_env == "APODEX_API_KEY" + assert apodex.api_base_env == "APODEX_API_BASE" + assert apodex.param_mappings["max_completion_tokens"] == "max_tokens" + assert apodex.supported_endpoints == ["/v1/chat/completions", "/v1/responses", "/v1/messages"] + assert JSONProviderRegistry.supports_responses_api("apodex") is True + + def test_provider_resolution(self): + model, provider, _, api_base = litellm.get_llm_provider(model=CORE_MODEL) + assert (model, provider, api_base) == ("apodex-1.1", "apodex", "https://api.apodex.ai/v1") + + def test_api_base_autodetection(self): + _, provider, api_key, _ = litellm.get_llm_provider(model="apodex-1.1", api_base="https://api.apodex.ai/v1") + assert provider == "apodex" + assert api_key == "sk-apodex-test" + + def test_explicit_api_base_and_key_win(self): + _, provider, api_key, api_base = litellm.get_llm_provider( + model=CORE_MODEL, api_base="https://gateway.internal/v1", api_key="sk-override" + ) + assert (provider, api_key, api_base) == ("apodex", "sk-override", "https://gateway.internal/v1") + + +class TestApodexStreamDefault: + """Apodex defaults `stream` to true, so a non-streaming call must pin it to false. + + Regression guard: the OpenAI SDK omits `stream` when it is false, which would make + litellm.completion() receive SSE and fail to parse it. + """ + + def test_chat_completion_pins_stream_false(self): + captured: dict = {} + response = litellm.completion( + model=CORE_MODEL, + messages=[{"role": "user", "content": "hi"}], + client=_openai_client(captured), + ) + + assert captured["url"] == "https://api.apodex.ai/v1/chat/completions" + assert captured["body"]["stream"] is False + assert captured["body"]["model"] == "apodex-1.1" + assert response.choices[0].message.reasoning_content == "let me think" + + def test_chat_completion_streaming_sends_stream_true(self): + captured: dict = {} + chunks = list( + litellm.completion( + model=CORE_MODEL, + messages=[{"role": "user", "content": "hi"}], + stream=True, + client=_openai_client(captured, stream=True), + ) + ) + + assert captured["body"]["stream"] is True + assert chunks + + def test_user_supplied_extra_body_is_preserved(self): + captured: dict = {} + litellm.completion( + model=CORE_MODEL, + messages=[{"role": "user", "content": "hi"}], + extra_body={"mcp_servers": [{"name": "docs", "url": "https://example.com/mcp"}]}, + client=_openai_client(captured), + ) + + assert captured["body"]["stream"] is False + assert captured["body"]["mcp_servers"] == [{"name": "docs", "url": "https://example.com/mcp"}] + + def test_responses_api_pins_stream_false(self): + captured: dict = {} + + class CapturingHandler(HTTPHandler): + def post(self, *args, **kwargs): + captured.update(url=kwargs.get("url"), body=kwargs.get("json")) + raise RuntimeError("captured") + + with pytest.raises(Exception): + litellm.responses(model=DEEP_RESEARCH_MODEL, input="hi", client=CapturingHandler()) + + assert captured["url"] == "https://api.apodex.ai/v1/responses" + assert captured["body"]["stream"] is False + assert captured["body"]["model"] == "apodex-1-1-deep-research" + + @pytest.mark.asyncio + async def test_responses_api_streaming_sends_stream_true(self): + captured: dict = {} + + class CapturingHandler(AsyncHTTPHandler): + async def post(self, *args, **kwargs): + captured.update(body=kwargs.get("json")) + raise RuntimeError("captured") + + with pytest.raises(Exception): + await litellm.aresponses(model=DEEP_RESEARCH_MODEL, input="hi", stream=True, client=CapturingHandler()) + + assert captured["body"]["stream"] is True + + def test_flag_is_opt_in_for_other_json_providers(self): + from litellm.llms.openai_like.json_loader import JSONProviderRegistry + + pinstripes = JSONProviderRegistry.get("pinstripes") + assert pinstripes is not None + assert "send_explicit_stream_false" not in pinstripes.special_handling + + config = ProviderConfigManager.get_provider_chat_config( + model="ps/glm-4.5-air", provider=LlmProviders.PINSTRIPES + ) + params = config.map_openai_params({}, {}, "ps/glm-4.5-air", False) + assert "stream" not in params + assert "stream" not in (params.get("extra_body") or {}) + + +class TestApodexToolSupport: + """Deep research tiers reject OpenAI-style tools; core models accept them.""" + + def test_deep_research_drops_tool_params(self): + config = ProviderConfigManager.get_provider_chat_config( + model="apodex-1-1-deep-research", provider=LlmProviders.APODEX + ) + supported = config.get_supported_openai_params("apodex-1-1-deep-research") + assert "tools" not in supported + assert "tool_choice" not in supported + + def test_core_model_keeps_tool_params(self): + config = ProviderConfigManager.get_provider_chat_config(model="apodex-1.1", provider=LlmProviders.APODEX) + supported = config.get_supported_openai_params("apodex-1.1") + assert "tools" in supported + assert "tool_choice" in supported + + def test_max_completion_tokens_maps_to_max_tokens(self): + config = ProviderConfigManager.get_provider_chat_config(model="apodex-1.1", provider=LlmProviders.APODEX) + params = config.map_openai_params({"max_completion_tokens": 512}, {}, "apodex-1.1", False) + assert params["max_tokens"] == 512 + assert "max_completion_tokens" not in params + + +class TestApodexAnthropicMessages: + """Apodex serves POST /v1/messages natively, so the payload is forwarded untranslated.""" + + def test_native_passthrough_config(self): + config = ProviderConfigManager.get_provider_anthropic_messages_config( + model="apodex-1.1", provider=LlmProviders.APODEX + ) + assert config is not None + assert type(config).__name__ == "JSONProviderAnthropicMessagesConfig" + assert ( + config.get_complete_url( + api_base=None, api_key=None, model="apodex-1.1", optional_params={}, litellm_params={} + ) + == "https://api.apodex.ai/v1/messages" + ) + + def test_headers_use_provider_api_key(self): + config = ProviderConfigManager.get_provider_anthropic_messages_config( + model="apodex-1.1", provider=LlmProviders.APODEX + ) + assert config is not None + headers, _ = config.validate_anthropic_messages_environment( + headers={}, model="apodex-1.1", messages=[], optional_params={}, litellm_params={} + ) + assert headers["authorization"] == "Bearer sk-apodex-test" + assert headers["anthropic-version"] == "2023-06-01" + + +class TestApodexModelMetadata: + @pytest.fixture(scope="class") + def model_cost(self) -> dict: + with open(REPO_ROOT / "model_prices_and_context_window.json") as f: + return json.load(f) + + def test_core_model_pricing(self, model_cost: dict): + info = model_cost["apodex/apodex-1.1"] + assert info["litellm_provider"] == "apodex" + assert info["mode"] == "chat" + assert info["max_input_tokens"] == 262144 + assert info["input_cost_per_token"] == 3e-07 + assert info["cache_read_input_token_cost"] == 3e-08 + assert info["output_cost_per_token"] == 3e-06 + # Requests over 200K input tokens are billed at 2x across every tier + assert info["input_cost_per_token_above_200k_tokens"] == 6e-07 + assert info["cache_read_input_token_cost_above_200k_tokens"] == 6e-08 + assert info["output_cost_per_token_above_200k_tokens"] == 6e-06 + assert info["supports_prompt_caching"] is True + assert info["supports_function_calling"] is True + assert info["supported_endpoints"] == ["/v1/chat/completions", "/v1/responses", "/v1/messages"] + + def test_deep_research_model_pricing(self, model_cost: dict): + info = model_cost["apodex/apodex-1-1-deep-research"] + assert info["max_input_tokens"] == 131072 + assert info["max_output_tokens"] == 65536 + assert info["input_cost_per_token"] == 5e-06 + assert info["output_cost_per_token"] == 2e-05 + assert info["supports_function_calling"] is False + assert info["supports_response_schema"] is False + assert info["supports_prompt_caching"] is False + assert info["supports_web_search"] is True + assert info["supported_endpoints"] == ["/v1/chat/completions", "/v1/responses"] + + def test_every_apodex_model_is_registered(self, model_cost: dict): + assert {key for key in model_cost if key.startswith("apodex/")} == { + "apodex/apodex-1.1", + "apodex/apodex-1.1-mini", + "apodex/apodex-1-1-deep-research", + "apodex/apodex-1-1-deep-solve", + "apodex/apodex-1-1-deep-discover", + "apodex/apodex-1-0-deep-research", + "apodex/apodex-1-0-deep-solve", + "apodex/apodex-1-0-deep-discover", + } + + def test_backup_cost_map_in_sync(self, model_cost: dict): + with open(REPO_ROOT / "litellm" / "model_prices_and_context_window_backup.json") as f: + backup = json.load(f) + for key in (key for key in model_cost if key.startswith("apodex/")): + assert backup[key] == model_cost[key], f"{key} differs between main and backup cost maps" + + def test_cost_tracks_cached_input_separately(self): + captured: dict = {} + response = litellm.completion( + model=CORE_MODEL, + messages=[{"role": "user", "content": "hi"}], + client=_openai_client(captured), + ) + + # 500 fresh input + 500 cached input + 100 output + expected = 500 * 3e-07 + 500 * 3e-08 + 100 * 3e-06 + assert litellm.completion_cost(response, model=CORE_MODEL) == pytest.approx(expected) diff --git a/tests/test_litellm/llms/openai_like/test_json_providers.py b/tests/test_litellm/llms/openai_like/test_json_providers.py index c8743e1809d..442dd554885 100644 --- a/tests/test_litellm/llms/openai_like/test_json_providers.py +++ b/tests/test_litellm/llms/openai_like/test_json_providers.py @@ -245,6 +245,63 @@ class TestPinstripes: assert result["temperature"] == 0.7 +class TestTemperatureConstraints: + """`constraints` in providers.json clamp temperature before the request is sent.""" + + @staticmethod + def _config(constraints: dict): + from litellm.llms.openai_like.dynamic_config import create_config_class + from litellm.llms.openai_like.json_loader import SimpleProviderConfig + + provider = SimpleProviderConfig( + "constrained", + { + "base_url": "https://api.constrained.test/v1", + "api_key_env": "CONSTRAINED_API_KEY", + "constraints": constraints, + }, + ) + return create_config_class(provider)() + + def test_temperature_clamped_to_max(self): + config = self._config({"temperature_max": 1.0}) + result = config.map_openai_params({"temperature": 1.8}, {}, "some-model", False) + assert result["temperature"] == 1.0 + + def test_temperature_clamped_to_min(self): + config = self._config({"temperature_min": 0.1}) + result = config.map_openai_params({"temperature": 0.0}, {}, "some-model", False) + assert result["temperature"] == 0.1 + + def test_temperature_within_range_is_untouched(self): + config = self._config({"temperature_min": 0.1, "temperature_max": 1.0}) + result = config.map_openai_params({"temperature": 0.7}, {}, "some-model", False) + assert result["temperature"] == 0.7 + + def test_temperature_floor_applies_only_when_n_gt_1(self): + config = self._config({"temperature_min_with_n_gt_1": 0.3}) + + single = config.map_openai_params({"temperature": 0.0, "n": 1}, {}, "some-model", False) + assert single["temperature"] == 0.0 + + multiple = config.map_openai_params({"temperature": 0.0, "n": 2}, {}, "some-model", False) + assert multiple["temperature"] == 0.3 + + def test_no_constraints_leaves_temperature_alone(self): + config = self._config({}) + result = config.map_openai_params({"temperature": 1.9}, {}, "some-model", False) + assert result["temperature"] == 1.9 + + def test_caller_optional_params_are_not_mutated(self): + config = self._config({"temperature_max": 1.0}) + optional_params = {"temperature": 1.8} + result = config.map_openai_params({"max_tokens": 10}, optional_params, "some-model", False) + + assert result["temperature"] == 1.0 + assert result["max_tokens"] == 10 + assert optional_params == {"temperature": 1.8} + + class TestDarkbloom: def test_darkbloom_json_config_exists(self): from litellm.llms.openai_like.json_loader import JSONProviderRegistry diff --git a/type-discipline-budget.json b/type-discipline-budget.json index 8e55b1533ea..b6752ada511 100644 --- a/type-discipline-budget.json +++ b/type-discipline-budget.json @@ -27,10 +27,10 @@ "limit": 0 }, "LIT010": { - "limit": 16715 + "limit": 16707 }, "LIT011": { - "limit": 5593 + "limit": 5589 }, "LIT012": { "limit": 4519 From 3b4ff9127b68ddcc446bec441e6b3de1e4d2d0d0 Mon Sep 17 00:00:00 2001 From: zhanghanduo Date: Sun, 16 Aug 2026 12:37:03 +0800 Subject: [PATCH 02/12] refactor(apodex): move to a Python provider with model-aware transformations The JSON provider path applies one contract to a whole provider, which is wrong for Apodex: its two model families take different parameters. Replaces the providers.json entry with litellm/llms/apodex/, reverting the shared openai_like machinery to its original state. /v1/responses is now keyed off the model. Core models are a stateless subset, so store is pinned false and previous_response_id / background are rejected rather than passed upstream to fail with a 400. The deep research tiers keep all three, so background survives a client disconnect. /v1/messages resolves per model too. Apodex serves the protocol natively for the core models only, so the deep research tiers get no native config and fall back to translation instead of hitting a path that does not serve them. Chat completions pin stream to false for both families, drop tool params on the deep research tiers, and rename max_completion_tokens to max_tokens. The responses config also stops inheriting OpenAI's OPENAI_API_KEY fallback, which would otherwise forward an unrelated OpenAI key to Apodex. Tests live under tests/test_litellm/llms/apodex/ and touch no existing test file. --- basedpyright-code-budget.json | 4 +- litellm/__init__.py | 10 + litellm/_lazy_imports_registry.py | 7 + litellm/constants.py | 2 +- .../get_llm_provider_logic.py | 10 +- litellm/llms/apodex/chat/transformation.py | 119 +++++++ litellm/llms/apodex/common_utils.py | 36 ++ .../llms/apodex/messages/transformation.py | 52 +++ .../llms/apodex/responses/transformation.py | 100 ++++++ litellm/llms/openai_like/README.md | 10 +- litellm/llms/openai_like/dynamic_config.py | 72 ++-- litellm/llms/openai_like/providers.json | 12 - litellm/utils.py | 14 + .../apodex/test_apodex_chat_transformation.py | 209 ++++++++++++ .../llms/apodex/test_apodex_common_utils.py | 136 ++++++++ .../test_apodex_messages_transformation.py | 98 ++++++ .../test_apodex_responses_transformation.py | 177 ++++++++++ .../llms/openai_like/test_apodex_provider.py | 310 ------------------ .../llms/openai_like/test_json_providers.py | 57 ---- type-discipline-budget.json | 4 +- 20 files changed, 1001 insertions(+), 438 deletions(-) create mode 100644 litellm/llms/apodex/chat/transformation.py create mode 100644 litellm/llms/apodex/common_utils.py create mode 100644 litellm/llms/apodex/messages/transformation.py create mode 100644 litellm/llms/apodex/responses/transformation.py create mode 100644 tests/test_litellm/llms/apodex/test_apodex_chat_transformation.py create mode 100644 tests/test_litellm/llms/apodex/test_apodex_common_utils.py create mode 100644 tests/test_litellm/llms/apodex/test_apodex_messages_transformation.py create mode 100644 tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py delete mode 100644 tests/test_litellm/llms/openai_like/test_apodex_provider.py diff --git a/basedpyright-code-budget.json b/basedpyright-code-budget.json index 0109927f2c6..06010c706e3 100644 --- a/basedpyright-code-budget.json +++ b/basedpyright-code-budget.json @@ -99,7 +99,7 @@ "limit": 0 }, "reportUnknownArgumentType": { - "limit": 44774 + "limit": 44776 }, "reportUnknownLambdaType": { "limit": 113 @@ -111,7 +111,7 @@ "limit": 19967 }, "reportUnknownVariableType": { - "limit": 30879 + "limit": 30881 }, "reportUnnecessaryCast": { "limit": 117 diff --git a/litellm/__init__.py b/litellm/__init__.py index 8961de940a0..1bd171fcc7b 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -638,6 +638,7 @@ snowflake_models: Set = set() gradient_ai_models: Set = set() llama_models: Set = set() nscale_models: Set = set() +apodex_models: Set = set() nebius_models: Set = set() nebius_embedding_models: Set = set() aiml_models: Set = set() @@ -828,6 +829,8 @@ def _populate_provider_model_sets(model_cost_map: Dict) -> None: llama_models.add(key) elif value.get("litellm_provider") == "nscale": nscale_models.add(key) + elif value.get("litellm_provider") == "apodex": + apodex_models.add(key) elif value.get("litellm_provider") == "azure_ai": azure_ai_models.add(key) elif value.get("litellm_provider") == "voyage": @@ -1052,6 +1055,7 @@ model_list = list( | llama_models | featherless_ai_models | nscale_models + | apodex_models | deepgram_models | elevenlabs_models | dashscope_models @@ -1156,6 +1160,7 @@ def _build_models_by_provider() -> dict: "gradient_ai": gradient_ai_models, "meta_llama": llama_models, "nscale": nscale_models, + "apodex": apodex_models, "featherless_ai": featherless_ai_models, "deepgram": deepgram_models, "elevenlabs": elevenlabs_models, @@ -1782,6 +1787,9 @@ if TYPE_CHECKING: from .llms.perplexity.responses.transformation import ( PerplexityResponsesConfig as PerplexityResponsesConfig, ) + from .llms.apodex.responses.transformation import ( + ApodexResponsesConfig as ApodexResponsesConfig, + ) from .llms.databricks.responses.transformation import ( DatabricksResponsesAPIConfig as DatabricksResponsesAPIConfig, ) @@ -1855,6 +1863,7 @@ if TYPE_CHECKING: PerplexityChatConfig as _PerplexityChatConfig, ) from .llms.nscale.chat.transformation import NscaleConfig as _NscaleConfig + from .llms.apodex.chat.transformation import ApodexChatConfig as _ApodexChatConfig from .llms.watsonx.chat.transformation import ( IBMWatsonXChatConfig as _IBMWatsonXChatConfig, ) @@ -1890,6 +1899,7 @@ if TYPE_CHECKING: AzureOpenAIO1Config: Type[_AzureOpenAIO1Config] PerplexityChatConfig: Type[_PerplexityChatConfig] NscaleConfig: Type[_NscaleConfig] + ApodexChatConfig: Type[_ApodexChatConfig] IBMWatsonXChatConfig: Type[_IBMWatsonXChatConfig] IBMWatsonXAIConfig: Type[_IBMWatsonXAIConfig] LiteLLMProxyChatConfig: Type[_LiteLLMProxyChatConfig] diff --git a/litellm/_lazy_imports_registry.py b/litellm/_lazy_imports_registry.py index 89c72acc06d..c8e0024ab44 100644 --- a/litellm/_lazy_imports_registry.py +++ b/litellm/_lazy_imports_registry.py @@ -238,6 +238,7 @@ LLM_CONFIG_NAMES: Final = ( "HostedVLLMResponsesAPIConfig", "VolcEngineResponsesAPIConfig", "PerplexityResponsesConfig", + "ApodexResponsesConfig", "DatabricksResponsesAPIConfig", "OpenRouterResponsesAPIConfig", "BedrockMantleResponsesAPIConfig", @@ -291,6 +292,7 @@ LLM_CONFIG_NAMES: Final = ( "LmStudioEmbeddingConfig", "NscaleConfig", "PerplexityChatConfig", + "ApodexChatConfig", "AzureOpenAIO1Config", "IBMWatsonXAIConfig", "IBMWatsonXChatConfig", @@ -961,6 +963,10 @@ _LLM_CONFIGS_IMPORT_MAP: Final = { ".llms.perplexity.responses.transformation", "PerplexityResponsesConfig", ), + "ApodexResponsesConfig": ( + ".llms.apodex.responses.transformation", + "ApodexResponsesConfig", + ), "DatabricksResponsesAPIConfig": ( ".llms.databricks.responses.transformation", "DatabricksResponsesAPIConfig", @@ -1110,6 +1116,7 @@ _LLM_CONFIGS_IMPORT_MAP: Final = { ".llms.perplexity.chat.transformation", "PerplexityChatConfig", ), + "ApodexChatConfig": (".llms.apodex.chat.transformation", "ApodexChatConfig"), "AzureOpenAIO1Config": ( ".llms.azure.chat.o_series_transformation", "AzureOpenAIO1Config", diff --git a/litellm/constants.py b/litellm/constants.py index 7c99b0c9c46..0736603ab4c 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -824,7 +824,7 @@ openai_compatible_providers: Final[list] = [ "pinstripes", # Pinstripes - JSON-configured provider "darkbloom", "meta", # Meta Model API (Muse Spark) - JSON-configured provider - "apodex", # Apodex - JSON-configured provider + "apodex", ] openai_text_completion_compatible_providers: Final[list] = [ # providers that support `/v1/completions` "together_ai", diff --git a/litellm/litellm_core_utils/get_llm_provider_logic.py b/litellm/litellm_core_utils/get_llm_provider_logic.py index 9f07710007f..ac1faeda645 100644 --- a/litellm/litellm_core_utils/get_llm_provider_logic.py +++ b/litellm/litellm_core_utils/get_llm_provider_logic.py @@ -349,9 +349,10 @@ def get_llm_provider( elif endpoint == "https://api.meta.ai/v1": custom_llm_provider = "meta" dynamic_api_key = get_secret_str("META_API_KEY") - elif endpoint == "https://api.apodex.ai/v1": - custom_llm_provider = "apodex" - dynamic_api_key = get_secret_str("APODEX_API_KEY") + elif endpoint == litellm.ApodexChatConfig.API_BASE_URL: + custom_llm_provider = "apodex" # rebind-ok: dispatch chain resolves in place + # rebind-ok: dispatch chain resolves in place + dynamic_api_key = litellm.ApodexChatConfig.get_api_key() if api_base is not None and not isinstance(api_base, str): raise Exception(f"api base needs to be a string. api_base={api_base}") @@ -757,6 +758,9 @@ def _get_openai_compatible_provider_info( api_base, dynamic_api_key, ) = litellm.NscaleConfig()._get_openai_compatible_provider_info(api_base=api_base, api_key=api_key) + elif custom_llm_provider == "apodex": + api_base = litellm.ApodexChatConfig.get_api_base(api_base) # rebind-ok: dispatch chain resolves in place + dynamic_api_key = litellm.ApodexChatConfig.get_api_key(api_key) # rebind-ok: resolved in place elif custom_llm_provider == "heroku": ( api_base, diff --git a/litellm/llms/apodex/chat/transformation.py b/litellm/llms/apodex/chat/transformation.py new file mode 100644 index 00000000000..300e3e57991 --- /dev/null +++ b/litellm/llms/apodex/chat/transformation.py @@ -0,0 +1,119 @@ +""" +Apodex chat completions — OpenAI-compatible, with two provider quirks: + +- `stream` defaults to true upstream, so a non-streaming call has to say so + explicitly or Apodex answers with SSE that a plain call cannot parse +- the Deep Research tiers ignore sampling parameters and reject OpenAI-style + tools; only the core models take them + +Ref: https://platform.apodex.ai/docs/chat-completions +""" + +from collections.abc import Mapping +from typing import Final + +from litellm.llms.openai.chat.gpt_transformation import OpenAIGPTConfig + +from ..common_utils import ( + APODEX_API_BASE_URL, + get_apodex_api_base, + get_apodex_api_key, + is_deep_research_model, +) + +_DEEP_RESEARCH_PARAMS: Final = ( + "max_tokens", + "max_completion_tokens", + "stream", + "stream_options", + "extra_headers", + "max_retries", +) + +_CORE_PARAMS: Final = ( + *_DEEP_RESEARCH_PARAMS, + "temperature", + "top_p", + "stop", + "seed", + "n", + "tools", + "tool_choice", + "function_call", + "functions", + "parallel_tool_calls", +) + + +class ApodexChatConfig(OpenAIGPTConfig): + """ + Reference: https://platform.apodex.ai/docs + API Key: APODEX_API_KEY + Default API Base: https://api.apodex.ai/v1 + """ + + API_BASE_URL = APODEX_API_BASE_URL + + @property + def custom_llm_provider(self) -> str | None: + return "apodex" + + @staticmethod + def get_api_key(api_key: str | None = None) -> str | None: + return get_apodex_api_key(api_key) + + @staticmethod + def get_api_base(api_base: str | None = None) -> str | None: + return get_apodex_api_base(api_base) + + def _get_openai_compatible_provider_info( + self, api_base: str | None, api_key: str | None + ) -> tuple[str | None, str | None]: + return get_apodex_api_base(api_base), get_apodex_api_key(api_key) + + def get_supported_openai_params(self, model: str) -> list: # mutable-ok: matches the base-class signature + supported: Final = _DEEP_RESEARCH_PARAMS if is_deep_research_model(model) else _CORE_PARAMS + return list(supported) # mutable-ok: matches the base-class signature + + def map_openai_params( + self, + non_default_params: dict, # mutable-ok: matches the base-class signature + optional_params: dict, # mutable-ok: matches the base-class signature + model: str, + drop_params: bool, + ) -> dict: # mutable-ok: matches the base-class signature + mapped: Final = super().map_openai_params( + non_default_params=non_default_params, + optional_params=optional_params, + model=model, + drop_params=drop_params, + ) + + # Apodex documents max_tokens only. + renamed: Final = ( + mapped + if "max_completion_tokens" not in mapped + else { # mutable-ok: JSON request body + **{ # mutable-ok: JSON request body + key: value for key, value in mapped.items() if key != "max_completion_tokens" + }, + "max_tokens": mapped["max_completion_tokens"], + } + ) + + if renamed.get("stream"): + return renamed + + # The OpenAI SDK drops `stream` from the body when it is false, which would + # leave Apodex on its streaming default. extra_body is merged into the + # request body by the SDK, so it survives that drop. + requested_extra_body: Final = renamed.get("extra_body") + extra_body: Final = ( + requested_extra_body + if isinstance(requested_extra_body, Mapping) + else {} # mutable-ok: JSON request body + ) + return { # mutable-ok: JSON request body + **renamed, + "extra_body": {"stream": False, **extra_body}, # mutable-ok: JSON request body + } diff --git a/litellm/llms/apodex/common_utils.py b/litellm/llms/apodex/common_utils.py new file mode 100644 index 00000000000..ad29673ff4b --- /dev/null +++ b/litellm/llms/apodex/common_utils.py @@ -0,0 +1,36 @@ +""" +Shared helpers for the Apodex provider. + +Apodex serves two model families on one base URL, and the model id picks which +contract applies. Core models (apodex-1.1, apodex-1.1-mini) are plain inference +with native sampling parameters. The Deep Research tiers run an agent that +plans, searches and iterates, so they ignore sampling parameters, reject +OpenAI-style tools, and keep server-side state. + +Ref: https://platform.apodex.ai/docs/models +""" + +from typing import Final + +from litellm.secret_managers.main import get_secret_str + +APODEX_API_BASE_URL: Final = "https://api.apodex.ai/v1" + +_DEEP_RESEARCH_MARKER: Final = "-deep-" + + +def strip_provider_prefix(model: str) -> str: + return model.rpartition("/")[2] + + +def is_deep_research_model(model: str) -> bool: + """True for the Deep Research / Solve / Discover tiers, e.g. apodex-1-1-deep-solve.""" + return _DEEP_RESEARCH_MARKER in strip_provider_prefix(model) + + +def get_apodex_api_key(api_key: str | None = None) -> str | None: + return api_key or get_secret_str("APODEX_API_KEY") + + +def get_apodex_api_base(api_base: str | None = None) -> str: + return api_base or get_secret_str("APODEX_API_BASE") or APODEX_API_BASE_URL diff --git a/litellm/llms/apodex/messages/transformation.py b/litellm/llms/apodex/messages/transformation.py new file mode 100644 index 00000000000..7788b0b5eb7 --- /dev/null +++ b/litellm/llms/apodex/messages/transformation.py @@ -0,0 +1,52 @@ +""" +Apodex Anthropic Messages — native passthrough for the core models only. + +Apodex implements the Anthropic protocol itself at POST /v1/messages and serves +the core models there, so the payload is forwarded untranslated and +Anthropic-only features such as `thinking` and `cache_control` survive. The Deep +Research tiers are not served on that path, so `ProviderConfigManager` hands back +no config for them and they fall back to LiteLLM's Anthropic-to-chat-completions +translation. + +Ref: https://platform.apodex.ai/docs/anthropic-messages +""" + +from litellm.llms.openai_like.messages.transformation import ( + OpenAILikeAnthropicMessagesConfig, +) + +from ..common_utils import get_apodex_api_base, get_apodex_api_key + + +class ApodexAnthropicMessagesConfig(OpenAILikeAnthropicMessagesConfig): + @property + def custom_llm_provider(self) -> str | None: + return "apodex" + + def should_strip_billing_metadata(self) -> bool: + return True + + def validate_anthropic_messages_environment( + self, + headers: dict[str, str], # mutable-ok: matches the base-class signature + model: str, + messages: list[object], # mutable-ok: matches the base-class signature + optional_params: dict, # mutable-ok: matches the base-class signature + litellm_params: dict, # mutable-ok: matches the base-class signature + api_key: str | None = None, + api_base: str | None = None, + ) -> tuple[dict[str, str], str | None]: # mutable-ok: matches the base-class signature + """Fill in the Apodex credentials and base URL. + + The returned api_base is what the handler hands to get_complete_url, so + resolving it here is enough to reach the native endpoint. + """ + return super().validate_anthropic_messages_environment( + headers=headers, + model=model, + messages=messages, + optional_params=optional_params, + litellm_params=litellm_params, + api_key=get_apodex_api_key(api_key), + api_base=get_apodex_api_base(api_base), + ) diff --git a/litellm/llms/apodex/responses/transformation.py b/litellm/llms/apodex/responses/transformation.py new file mode 100644 index 00000000000..04ca2e55c74 --- /dev/null +++ b/litellm/llms/apodex/responses/transformation.py @@ -0,0 +1,100 @@ +""" +Apodex Responses API — OpenAI-compatible, with a model-aware parameter contract. + +Apodex serves /v1/responses for both model families but they accept different +subsets, so the restrictions here are keyed off the model rather than applied +provider-wide: + +- core models are a stateless subset: `store` is forced to false, and + `previous_response_id` or `background` come back as HTTP 400 +- the Deep Research tiers keep server-side state, so they take all three +- both default `stream` to true, so a non-streaming call has to say so + +Ref: https://platform.apodex.ai/docs/responses-api + https://platform.apodex.ai/docs/models +""" + +from collections.abc import Mapping +from typing import Final + +import litellm +from litellm.llms.openai.responses.transformation import OpenAIResponsesAPIConfig +from litellm.types.llms.openai import ResponsesAPIOptionalRequestParams +from litellm.types.router import GenericLiteLLMParams +from litellm.types.utils import LlmProviders + +from ..common_utils import get_apodex_api_key, is_deep_research_model + +# Rejected by the core models with HTTP 400: there is no server-side conversation +# to resume and requests are always executed inline. +_STATEFUL_PARAMS: Final = ("previous_response_id", "background") + + +class ApodexResponsesConfig(OpenAIResponsesAPIConfig): + @property + def custom_llm_provider(self) -> LlmProviders: + return LlmProviders.APODEX + + def validate_environment( + self, + headers: dict, # mutable-ok: matches the base-class signature + model: str, + litellm_params: GenericLiteLLMParams | None, + ) -> dict: # mutable-ok: matches the base-class signature + """Resolve the Apodex key rather than inheriting OpenAI's OPENAI_API_KEY fallback, + which would otherwise forward an unrelated OpenAI key to Apodex.""" + resolved_params: Final = litellm_params or GenericLiteLLMParams() + api_key: Final = get_apodex_api_key(resolved_params.api_key) + if api_key is None: + return headers + return { # mutable-ok: matches the base-class signature + **headers, + "Content-Type": "application/json", + "Authorization": f"Bearer {api_key}", + } + + def get_supported_openai_params(self, model: str) -> list: # mutable-ok: matches the base-class signature + inherited: Final = super().get_supported_openai_params(model) + if is_deep_research_model(model): + return inherited + return [ # mutable-ok: matches the base-class signature + param for param in inherited if param not in _STATEFUL_PARAMS + ] + + def map_openai_params( + self, + response_api_optional_params: ResponsesAPIOptionalRequestParams, + model: str, + drop_params: bool, + ) -> dict: # mutable-ok: matches the base-class signature + mapped: Final = super().map_openai_params( + response_api_optional_params=response_api_optional_params, + model=model, + drop_params=drop_params, + ) + + stateless: Final = ( + mapped + if is_deep_research_model(model) + else self._enforce_stateless(mapped, model=model, drop_params=drop_params) + ) + + if stateless.get("stream"): + return {**stateless} # mutable-ok: JSON request body + return {**stateless, "stream": False} # mutable-ok: JSON request body + + @staticmethod + def _enforce_stateless(params: Mapping[str, object], model: str, drop_params: bool) -> Mapping[str, object]: + """Core models only: drop what the stateless subset rejects and pin store to false.""" + if params.get("store") is True and not (drop_params or litellm.drop_params): + raise litellm.UnsupportedParamsError( + message=( + f"apodex model {model} does not support store=True on /v1/responses: the endpoint is a " + "stateless subset. To drop this, set `litellm.drop_params = True`" + ), + status_code=400, + ) + kept: Final = { # mutable-ok: JSON request body + key: value for key, value in params.items() if key not in _STATEFUL_PARAMS + } + return {**kept, "store": False} # mutable-ok: JSON request body diff --git a/litellm/llms/openai_like/README.md b/litellm/llms/openai_like/README.md index fc9375f2efb..e9aaafe48a1 100644 --- a/litellm/llms/openai_like/README.md +++ b/litellm/llms/openai_like/README.md @@ -59,15 +59,7 @@ That's it! The provider will be automatically loaded and available. // Optional: Special handling flags "special_handling": { - "convert_content_list_to_string": true, - - // Send "stream": false explicitly instead of omitting it. Needed by - // providers whose /v1/chat/completions and /v1/responses default to - // streaming, where omitting the field returns SSE to a non-streaming call - "send_explicit_stream_false": true, - - // Always send "store": false on /v1/responses - "force_store_false": true + "convert_content_list_to_string": true } } } diff --git a/litellm/llms/openai_like/dynamic_config.py b/litellm/llms/openai_like/dynamic_config.py index 52168e8bb66..19e29bcdcb2 100644 --- a/litellm/llms/openai_like/dynamic_config.py +++ b/litellm/llms/openai_like/dynamic_config.py @@ -2,7 +2,7 @@ Dynamic configuration class generator for JSON-based providers. """ -from collections.abc import Coroutine, Mapping +from collections.abc import Coroutine from typing import Any, Final, Literal, overload from litellm._logging import verbose_logger @@ -17,17 +17,6 @@ from litellm.types.llms.openai import AllMessageValues from .json_loader import SimpleProviderConfig -def _clamp_temperature(temperature: float, n: int, constraints: Mapping[str, float]) -> float: - capped: Final = ( - min(temperature, constraints["temperature_max"]) if "temperature_max" in constraints else temperature - ) - floored: Final = max(capped, constraints["temperature_min"]) if "temperature_min" in constraints else capped - floor_for_multiple_choices: Final = constraints.get("temperature_min_with_n_gt_1") - if n > 1 and floor_for_multiple_choices is not None: - return max(floored, floor_for_multiple_choices) - return floored - - def create_config_class(provider: SimpleProviderConfig): """Generate config class dynamically from JSON configuration""" @@ -142,36 +131,37 @@ def create_config_class(provider: SimpleProviderConfig): """Apply parameter mappings and constraints""" supported_params: Final = self.get_supported_openai_params(model) - mapped: Final = { - **optional_params, - **{ - provider.param_mappings.get(param, param): value - for param, value in non_default_params.items() - if param in provider.param_mappings or param in supported_params - }, - } - constrained: Final = ( - mapped - if "temperature" not in mapped - else { - **mapped, - "temperature": _clamp_temperature( - temperature=mapped["temperature"], - n=mapped.get("n", 1), - constraints=provider.constraints, - ), - } - ) + # Apply supported params + for param, value in non_default_params.items(): + # Check parameter mappings first + if param in provider.param_mappings: + optional_params[provider.param_mappings[param]] = value + elif param in supported_params: + optional_params[param] = value - # The OpenAI SDK omits `stream` entirely when it is false, which makes - # stream-by-default providers answer a non-streaming call with SSE. Pin it - # on the wire through extra_body, which the SDK merges into the request body. - if not provider.special_handling.get("send_explicit_stream_false") or constrained.get("stream"): - return constrained - requested_extra_body: Final = constrained.get("extra_body") - extra_body: Final[dict] = requested_extra_body if isinstance(requested_extra_body, dict) else {} - return {**constrained, "extra_body": {"stream": False, **extra_body}} + # Apply temperature constraints if present + if "temperature" in optional_params: + temp = optional_params["temperature"] + constraints: Final = provider.constraints + + # Clamp to max + if "temperature_max" in constraints: + temp = min(temp, constraints["temperature_max"]) + + # Clamp to min + if "temperature_min" in constraints: + temp = max(temp, constraints["temperature_min"]) + + # Special case: temperature_min_with_n_gt_1 + if "temperature_min_with_n_gt_1" in constraints: + n: Final = optional_params.get("n", 1) + if n > 1 and temp < constraints["temperature_min_with_n_gt_1"]: + temp = constraints["temperature_min_with_n_gt_1"] + + optional_params["temperature"] = temp + + return optional_params @property def custom_llm_provider(self) -> str | None: @@ -242,8 +232,6 @@ def create_responses_config_class(provider: SimpleProviderConfig): ) -> dict: if provider.special_handling.get("force_store_false"): response_api_optional_request_params["store"] = False - if provider.special_handling.get("send_explicit_stream_false"): - response_api_optional_request_params.setdefault("stream", False) return super().transform_responses_api_request( model=model, input=input, diff --git a/litellm/llms/openai_like/providers.json b/litellm/llms/openai_like/providers.json index 18cf49e6ccc..164100d4194 100644 --- a/litellm/llms/openai_like/providers.json +++ b/litellm/llms/openai_like/providers.json @@ -175,18 +175,6 @@ "base_class": "openai_gpt", "supported_endpoints": ["/v1/chat/completions", "/v1/responses", "/v1/messages"] }, - "apodex": { - "base_url": "https://api.apodex.ai/v1", - "api_key_env": "APODEX_API_KEY", - "api_base_env": "APODEX_API_BASE", - "param_mappings": { - "max_completion_tokens": "max_tokens" - }, - "supported_endpoints": ["/v1/chat/completions", "/v1/responses", "/v1/messages"], - "special_handling": { - "send_explicit_stream_false": true - } - }, "pinstripes": { "base_url": "https://pinstripes.io/v1", "api_key_env": "PINSTRIPES_API_KEY", diff --git a/litellm/utils.py b/litellm/utils.py index d91d3092624..af6bc5f5867 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -7928,6 +7928,7 @@ class ProviderConfigManager: ), LlmProviders.GRADIENT_AI: (lambda: litellm.GradientAIConfig(), False), LlmProviders.NSCALE: (lambda: litellm.NscaleConfig(), False), + LlmProviders.APODEX: (lambda: litellm.ApodexChatConfig(), False), LlmProviders.HEROKU: (lambda: litellm.HerokuChatConfig(), False), LlmProviders.OCI: (lambda: litellm.OCIChatConfig(), False), LlmProviders.HYPERBOLIC: (lambda: litellm.HyperbolicChatConfig(), False), @@ -8255,6 +8256,17 @@ class ProviderConfigManager: ) return DeepSeekAnthropicMessagesConfig() + elif litellm.LlmProviders.APODEX == provider: + from litellm.llms.apodex.common_utils import is_deep_research_model + from litellm.llms.apodex.messages.transformation import ( + ApodexAnthropicMessagesConfig, + ) + + # Apodex only serves the core models on its native /v1/messages path; the + # deep research tiers get no config so they fall back to translation. + if is_deep_research_model(model): + return None + return ApodexAnthropicMessagesConfig() elif litellm.LlmProviders.TENCENT == provider: from litellm.llms.tencent.messages.transformation import ( TencentAnthropicMessagesConfig, @@ -8440,6 +8452,8 @@ class ProviderConfigManager: return litellm.ManusResponsesAPIConfig() elif litellm.LlmProviders.PERPLEXITY == provider: return litellm.PerplexityResponsesConfig() + elif litellm.LlmProviders.APODEX == provider: + return litellm.ApodexResponsesConfig() elif litellm.LlmProviders.DATABRICKS == provider: # Databricks Responses API is only compatible with OpenAI GPT models if model and "gpt" in model.lower(): diff --git a/tests/test_litellm/llms/apodex/test_apodex_chat_transformation.py b/tests/test_litellm/llms/apodex/test_apodex_chat_transformation.py new file mode 100644 index 00000000000..61fccf2483a --- /dev/null +++ b/tests/test_litellm/llms/apodex/test_apodex_chat_transformation.py @@ -0,0 +1,209 @@ +""" +Apodex chat completions transformation. +""" + +import json + +import httpx +import openai +import pytest + +import litellm +from litellm.types.utils import LlmProviders +from litellm.utils import ProviderConfigManager + +CORE_MODEL = "apodex/apodex-1.1" +DEEP_RESEARCH_MODEL = "apodex/apodex-1-1-deep-research" + +CHAT_RESPONSE = { + "id": "chatcmpl-abc123", + "object": "chat.completion", + "created": 1712345678, + "model": "apodex-1.1", + "choices": [ + { + "index": 0, + "message": {"role": "assistant", "content": "ok", "reasoning_content": "let me think"}, + "finish_reason": "stop", + } + ], + "usage": { + "prompt_tokens": 1000, + "completion_tokens": 100, + "total_tokens": 1100, + "prompt_tokens_details": {"cached_tokens": 500}, + }, +} + +STREAM_BODY = ( + b'data: {"id":"chatcmpl-abc123","object":"chat.completion.chunk","created":1,"model":"apodex-1.1",' + b'"choices":[{"index":0,"delta":{"content":"ok"},"finish_reason":null}]}\n\n' + b"data: [DONE]\n\n" +) + + +@pytest.fixture(autouse=True) +def _apodex_env(monkeypatch: pytest.MonkeyPatch): + """Resolve models against the in-repo cost map, not the published one.""" + monkeypatch.setenv("APODEX_API_KEY", "sk-apodex-test") + monkeypatch.delenv("APODEX_API_BASE", raising=False) + monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") + monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) + yield + + +def _client(captured: dict, *, stream: bool = False) -> openai.OpenAI: + def handler(request: httpx.Request) -> httpx.Response: + captured["url"] = str(request.url) + captured["body"] = json.loads(request.content) + if stream: + return httpx.Response(200, headers={"content-type": "text/event-stream"}, content=STREAM_BODY) + return httpx.Response(200, json=CHAT_RESPONSE) + + return openai.OpenAI( + api_key="sk-apodex-test", + base_url="https://api.apodex.ai/v1", + http_client=httpx.Client(transport=httpx.MockTransport(handler)), + ) + + +def _chat_config(model: str): + return ProviderConfigManager.get_provider_chat_config(model=model, provider=LlmProviders.APODEX) + + +class TestProviderResolution: + def test_prefixed_model_resolves_to_the_default_base(self): + model, provider, api_key, api_base = litellm.get_llm_provider(model=CORE_MODEL) + assert (model, provider, api_key, api_base) == ( + "apodex-1.1", + "apodex", + "sk-apodex-test", + "https://api.apodex.ai/v1", + ) + + def test_api_base_autodetection(self): + _, provider, api_key, _ = litellm.get_llm_provider(model="apodex-1.1", api_base="https://api.apodex.ai/v1") + assert provider == "apodex" + assert api_key == "sk-apodex-test" + + def test_explicit_api_base_and_key_win(self): + _, provider, api_key, api_base = litellm.get_llm_provider( + model=CORE_MODEL, api_base="https://gateway.internal/v1", api_key="sk-override" + ) + assert (provider, api_key, api_base) == ("apodex", "sk-override", "https://gateway.internal/v1") + + def test_api_base_env_override(self, monkeypatch: pytest.MonkeyPatch): + monkeypatch.setenv("APODEX_API_BASE", "https://env.apodex.test/v1") + _, _, _, api_base = litellm.get_llm_provider(model=CORE_MODEL) + assert api_base == "https://env.apodex.test/v1" + + +class TestStreamDefault: + """Apodex defaults `stream` to true, so a non-streaming call has to pin it to false. + + Regression guard: the OpenAI SDK drops `stream` from the body when it is false, + which would leave Apodex streaming SSE at a call that cannot parse it. + """ + + def test_non_streaming_call_pins_stream_false(self): + captured: dict = {} + response = litellm.completion( + model=CORE_MODEL, + messages=[{"role": "user", "content": "hi"}], + client=_client(captured), + ) + + assert captured["url"] == "https://api.apodex.ai/v1/chat/completions" + assert captured["body"]["stream"] is False + assert captured["body"]["model"] == "apodex-1.1" + assert response.choices[0].message.reasoning_content == "let me think" + + def test_streaming_call_sends_stream_true(self): + captured: dict = {} + chunks = list( + litellm.completion( + model=CORE_MODEL, + messages=[{"role": "user", "content": "hi"}], + stream=True, + client=_client(captured, stream=True), + ) + ) + + assert captured["body"]["stream"] is True + assert chunks + + def test_deep_research_models_pin_stream_too(self): + captured: dict = {} + litellm.completion( + model=DEEP_RESEARCH_MODEL, + messages=[{"role": "user", "content": "hi"}], + client=_client(captured), + ) + assert captured["body"]["stream"] is False + + def test_user_supplied_extra_body_is_preserved(self): + """Deep research tiers reach external tools through `mcp_servers` in extra_body.""" + captured: dict = {} + mcp_servers = [{"name": "docs", "url": "https://example.com/mcp"}] + litellm.completion( + model=DEEP_RESEARCH_MODEL, + messages=[{"role": "user", "content": "hi"}], + extra_body={"mcp_servers": mcp_servers}, + client=_client(captured), + ) + + assert captured["body"]["stream"] is False + assert captured["body"]["mcp_servers"] == mcp_servers + + +class TestSupportedParams: + def test_core_models_support_tools(self): + supported = _chat_config("apodex-1.1").get_supported_openai_params("apodex-1.1") + assert "tools" in supported + assert "tool_choice" in supported + assert "temperature" in supported + assert "top_p" in supported + + def test_deep_research_rejects_tools_and_sampling_params(self): + """The tiers document tools as unsupported and sampling params as ignored.""" + supported = _chat_config(DEEP_RESEARCH_MODEL).get_supported_openai_params("apodex-1-1-deep-research") + for param in ("tools", "tool_choice", "function_call", "functions", "parallel_tool_calls"): + assert param not in supported + assert "temperature" not in supported + assert "top_p" not in supported + assert "max_tokens" in supported + + def test_tools_on_a_deep_research_model_raise(self): + with pytest.raises(litellm.UnsupportedParamsError, match="tools"): + litellm.completion( + model=DEEP_RESEARCH_MODEL, + messages=[{"role": "user", "content": "hi"}], + tools=[{"type": "function", "function": {"name": "f", "parameters": {}}}], + client=_client({}), + ) + + def test_max_completion_tokens_is_renamed_to_max_tokens(self): + captured: dict = {} + litellm.completion( + model=CORE_MODEL, + messages=[{"role": "user", "content": "hi"}], + max_completion_tokens=512, + client=_client(captured), + ) + + assert captured["body"]["max_tokens"] == 512 + assert "max_completion_tokens" not in captured["body"] + + +class TestCostTracking: + def test_cached_input_is_billed_at_the_lower_rate(self): + captured: dict = {} + response = litellm.completion( + model=CORE_MODEL, + messages=[{"role": "user", "content": "hi"}], + client=_client(captured), + ) + + # 500 fresh input + 500 cached input + 100 output + expected = 500 * 3e-07 + 500 * 3e-08 + 100 * 3e-06 + assert litellm.completion_cost(response, model=CORE_MODEL) == pytest.approx(expected) diff --git a/tests/test_litellm/llms/apodex/test_apodex_common_utils.py b/tests/test_litellm/llms/apodex/test_apodex_common_utils.py new file mode 100644 index 00000000000..1bb72ca13d6 --- /dev/null +++ b/tests/test_litellm/llms/apodex/test_apodex_common_utils.py @@ -0,0 +1,136 @@ +""" +Apodex provider registration and model-family classification. +""" + +import json +from pathlib import Path + +import pytest + +import litellm +from litellm.llms.apodex.common_utils import ( + APODEX_API_BASE_URL, + get_apodex_api_base, + get_apodex_api_key, + is_deep_research_model, +) +from litellm.types.utils import LlmProviders + +REPO_ROOT = Path(__file__).parents[4] + +CORE_MODELS = ("apodex-1.1", "apodex-1.1-mini") +DEEP_RESEARCH_MODELS = ( + "apodex-1-1-deep-research", + "apodex-1-1-deep-solve", + "apodex-1-1-deep-discover", + "apodex-1-0-deep-research", + "apodex-1-0-deep-solve", + "apodex-1-0-deep-discover", +) + + +class TestModelFamily: + """The model id, not the provider, selects which Apodex contract applies.""" + + @pytest.mark.parametrize("model", CORE_MODELS) + def test_core_models_are_not_deep_research(self, model: str): + assert is_deep_research_model(model) is False + assert is_deep_research_model(f"apodex/{model}") is False + + @pytest.mark.parametrize("model", DEEP_RESEARCH_MODELS) + def test_deep_research_models_are_detected(self, model: str): + assert is_deep_research_model(model) is True + assert is_deep_research_model(f"apodex/{model}") is True + + def test_prefix_does_not_leak_into_classification(self): + """A provider prefix containing the marker must not flip a core model.""" + assert is_deep_research_model("some-deep-gateway/apodex-1.1") is False + + +class TestCredentialResolution: + def test_defaults(self, monkeypatch: pytest.MonkeyPatch): + monkeypatch.delenv("APODEX_API_BASE", raising=False) + monkeypatch.setenv("APODEX_API_KEY", "sk-env") + + assert get_apodex_api_base(None) == APODEX_API_BASE_URL + assert get_apodex_api_key(None) == "sk-env" + + def test_explicit_values_win_over_env(self, monkeypatch: pytest.MonkeyPatch): + monkeypatch.setenv("APODEX_API_BASE", "https://env.apodex.test/v1") + monkeypatch.setenv("APODEX_API_KEY", "sk-env") + + assert get_apodex_api_base("https://explicit.apodex.test/v1") == "https://explicit.apodex.test/v1" + assert get_apodex_api_key("sk-explicit") == "sk-explicit" + + def test_env_base_overrides_default(self, monkeypatch: pytest.MonkeyPatch): + monkeypatch.setenv("APODEX_API_BASE", "https://env.apodex.test/v1") + assert get_apodex_api_base(None) == "https://env.apodex.test/v1" + + +class TestRegistration: + def test_provider_enum_and_lists(self): + assert LlmProviders.APODEX.value == "apodex" + assert "apodex" in litellm.provider_list + assert "apodex" in litellm.constants.openai_compatible_providers + assert APODEX_API_BASE_URL in litellm.constants.openai_compatible_endpoints + + def test_not_registered_as_a_json_provider(self): + """Apodex needs model-aware transformations, so it must not fall into the + generic JSON path, which would shadow the Python configs in provider resolution.""" + from litellm.llms.openai_like.json_loader import JSONProviderRegistry + + assert JSONProviderRegistry.exists("apodex") is False + + def test_config_classes_resolve_from_the_lazy_registry(self): + assert litellm.ApodexChatConfig().custom_llm_provider == "apodex" + assert litellm.ApodexResponsesConfig().custom_llm_provider == LlmProviders.APODEX + + +class TestModelMetadata: + @pytest.fixture(scope="class") + def model_cost(self) -> dict: + with open(REPO_ROOT / "model_prices_and_context_window.json") as f: + return json.load(f) + + def test_every_apodex_model_is_registered(self, model_cost: dict): + assert {key for key in model_cost if key.startswith("apodex/")} == { + f"apodex/{model}" for model in (*CORE_MODELS, *DEEP_RESEARCH_MODELS) + } + + def test_core_model_pricing(self, model_cost: dict): + info = model_cost["apodex/apodex-1.1"] + assert info["litellm_provider"] == "apodex" + assert info["mode"] == "chat" + assert info["max_input_tokens"] == 262144 + assert info["input_cost_per_token"] == 3e-07 + assert info["cache_read_input_token_cost"] == 3e-08 + assert info["output_cost_per_token"] == 3e-06 + # Requests over 200K input tokens are billed at 2x across every tier + assert info["input_cost_per_token_above_200k_tokens"] == 6e-07 + assert info["cache_read_input_token_cost_above_200k_tokens"] == 6e-08 + assert info["output_cost_per_token_above_200k_tokens"] == 6e-06 + assert info["supports_prompt_caching"] is True + assert info["supports_function_calling"] is True + assert info["supported_endpoints"] == ["/v1/chat/completions", "/v1/responses", "/v1/messages"] + + def test_deep_research_model_pricing(self, model_cost: dict): + info = model_cost["apodex/apodex-1-1-deep-research"] + assert info["max_input_tokens"] == 131072 + assert info["max_output_tokens"] == 65536 + assert info["input_cost_per_token"] == 5e-06 + assert info["output_cost_per_token"] == 2e-05 + assert info["supports_function_calling"] is False + assert info["supports_response_schema"] is False + assert info["supports_prompt_caching"] is False + assert info["supports_web_search"] is True + + @pytest.mark.parametrize("model", DEEP_RESEARCH_MODELS) + def test_deep_research_models_are_not_on_the_native_messages_path(self, model_cost: dict, model: str): + """Apodex serves /v1/messages for the core models only.""" + assert "/v1/messages" not in model_cost[f"apodex/{model}"]["supported_endpoints"] + + def test_backup_cost_map_in_sync(self, model_cost: dict): + with open(REPO_ROOT / "litellm" / "model_prices_and_context_window_backup.json") as f: + backup = json.load(f) + for key in (key for key in model_cost if key.startswith("apodex/")): + assert backup[key] == model_cost[key], f"{key} differs between main and backup cost maps" diff --git a/tests/test_litellm/llms/apodex/test_apodex_messages_transformation.py b/tests/test_litellm/llms/apodex/test_apodex_messages_transformation.py new file mode 100644 index 00000000000..8be810539e3 --- /dev/null +++ b/tests/test_litellm/llms/apodex/test_apodex_messages_transformation.py @@ -0,0 +1,98 @@ +""" +Apodex Anthropic Messages transformation. + +Apodex implements the Anthropic protocol natively at POST /v1/messages, but only +serves the core models there. The Deep Research tiers must keep working on the +same route through LiteLLM's translation instead of being handed to a path that +would reject them. +""" + +import pytest + +import litellm +from litellm.types.utils import LlmProviders +from litellm.utils import ProviderConfigManager + +CORE_MODELS = ("apodex-1.1", "apodex-1.1-mini") +DEEP_RESEARCH_MODELS = ( + "apodex-1-1-deep-research", + "apodex-1-1-deep-solve", + "apodex-1-1-deep-discover", + "apodex-1-0-deep-research", + "apodex-1-0-deep-solve", + "apodex-1-0-deep-discover", +) + + +@pytest.fixture(autouse=True) +def _apodex_env(monkeypatch: pytest.MonkeyPatch): + monkeypatch.setenv("APODEX_API_KEY", "sk-apodex-test") + monkeypatch.delenv("APODEX_API_BASE", raising=False) + yield + + +def _messages_config(model: str): + return ProviderConfigManager.get_provider_anthropic_messages_config(model=model, provider=LlmProviders.APODEX) + + +def _complete_url() -> str: + """Resolve the endpoint the way the handler does: validate first, then build the URL. + + validate_anthropic_messages_environment returns the api_base the handler feeds + into get_complete_url, so the two steps have to run in that order. + """ + config = _messages_config("apodex-1.1") + assert config is not None + _, api_base = config.validate_anthropic_messages_environment( + headers={}, model="apodex-1.1", messages=[], optional_params={}, litellm_params={} + ) + return config.get_complete_url( + api_base=api_base, api_key=None, model="apodex-1.1", optional_params={}, litellm_params={} + ) + + +class TestNativePassthroughRouting: + @pytest.mark.parametrize("model", CORE_MODELS) + def test_core_models_get_the_native_config(self, model: str): + config = _messages_config(model) + assert config is not None + assert type(config).__name__ == "ApodexAnthropicMessagesConfig" + assert config.custom_llm_provider == "apodex" + + @pytest.mark.parametrize("model", DEEP_RESEARCH_MODELS) + def test_deep_research_models_fall_back_to_translation(self, model: str): + """No native config means LiteLLM translates to chat completions, which works, + instead of forwarding to a path Apodex does not serve for these tiers.""" + assert _messages_config(model) is None + + +class TestNativePassthroughRequest: + def test_url_targets_the_native_messages_path(self): + assert _complete_url() == "https://api.apodex.ai/v1/messages" + + def test_url_honours_an_api_base_override(self, monkeypatch: pytest.MonkeyPatch): + monkeypatch.setenv("APODEX_API_BASE", "https://env.apodex.test/v1") + assert _complete_url() == "https://env.apodex.test/v1/messages" + + def test_headers_use_the_provider_api_key(self): + config = _messages_config("apodex-1.1") + assert config is not None + headers, _ = config.validate_anthropic_messages_environment( + headers={}, model="apodex-1.1", messages=[], optional_params={}, litellm_params={} + ) + assert headers["authorization"] == "Bearer sk-apodex-test" + assert headers["anthropic-version"] == "2023-06-01" + assert headers["content-type"] == "application/json" + + def test_caller_supplied_auth_header_is_not_overwritten(self): + config = _messages_config("apodex-1.1") + assert config is not None + headers, _ = config.validate_anthropic_messages_environment( + headers={"x-api-key": "sk-caller"}, + model="apodex-1.1", + messages=[], + optional_params={}, + litellm_params={}, + ) + assert headers["x-api-key"] == "sk-caller" + assert "authorization" not in headers diff --git a/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py b/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py new file mode 100644 index 00000000000..d343f4553c8 --- /dev/null +++ b/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py @@ -0,0 +1,177 @@ +""" +Apodex Responses API transformation. + +The core models expose a stateless subset of /v1/responses while the Deep +Research tiers keep server-side state, so the parameter contract is keyed off +the model rather than applied provider-wide. +""" + +import pytest + +import litellm +from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler +from litellm.types.utils import LlmProviders +from litellm.utils import ProviderConfigManager + +CORE_MODEL = "apodex/apodex-1.1" +CORE_MINI_MODEL = "apodex/apodex-1.1-mini" +DEEP_RESEARCH_MODEL = "apodex/apodex-1-1-deep-research" + + +@pytest.fixture(autouse=True) +def _apodex_env(monkeypatch: pytest.MonkeyPatch): + monkeypatch.setenv("APODEX_API_KEY", "sk-apodex-test") + monkeypatch.delenv("APODEX_API_BASE", raising=False) + monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") + monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) + monkeypatch.setattr(litellm, "drop_params", False) + yield + + +_SENTINEL = "apodex-request-captured" + + +def _capture(**kwargs) -> dict: + """Run litellm.responses() and return the request it would have sent. + + Validation errors raised before the request is built propagate to the caller. + """ + captured: dict = {} + + class CapturingHandler(HTTPHandler): + def post(self, *args, **post_kwargs): + captured.update(url=post_kwargs.get("url"), body=post_kwargs.get("json")) + raise RuntimeError(_SENTINEL) + + try: + litellm.responses(client=CapturingHandler(), **kwargs) + except Exception as exc: + if _SENTINEL not in str(exc): + raise + assert captured, "no request was sent" + return captured + + +def _responses_config(model: str): + return ProviderConfigManager.get_provider_responses_api_config(model=model, provider=LlmProviders.APODEX) + + +class TestConfigSelection: + def test_python_config_is_used_for_every_apodex_model(self): + for model in ("apodex-1.1", "apodex-1.1-mini", "apodex-1-1-deep-research"): + config = _responses_config(model) + assert type(config).__name__ == "ApodexResponsesConfig" + + def test_auth_uses_the_apodex_key(self): + config = _responses_config("apodex-1.1") + assert config.validate_environment(headers={}, model="apodex-1.1", litellm_params=None) == { + "Content-Type": "application/json", + "Authorization": "Bearer sk-apodex-test", + } + + def test_auth_does_not_fall_back_to_an_openai_key(self, monkeypatch: pytest.MonkeyPatch): + """The inherited OpenAI config would forward OPENAI_API_KEY to Apodex.""" + monkeypatch.delenv("APODEX_API_KEY", raising=False) + monkeypatch.setenv("OPENAI_API_KEY", "sk-openai-must-not-leak") + + config = _responses_config("apodex-1.1") + assert config.validate_environment(headers={}, model="apodex-1.1", litellm_params=None) == {} + + def test_request_targets_the_apodex_responses_url(self): + assert _capture(model=CORE_MODEL, input="hi")["url"] == "https://api.apodex.ai/v1/responses" + + def test_request_honours_an_api_base_override(self, monkeypatch: pytest.MonkeyPatch): + monkeypatch.setenv("APODEX_API_BASE", "https://env.apodex.test/v1") + assert _capture(model=CORE_MODEL, input="hi")["url"] == "https://env.apodex.test/v1/responses" + + +class TestStreamDefault: + def test_non_streaming_pins_stream_false(self): + captured = _capture(model=DEEP_RESEARCH_MODEL, input="hi") + assert captured["url"] == "https://api.apodex.ai/v1/responses" + assert captured["body"]["stream"] is False + + @pytest.mark.asyncio + async def test_streaming_sends_stream_true(self): + captured: dict = {} + + class CapturingHandler(AsyncHTTPHandler): + async def post(self, *args, **kwargs): + captured.update(body=kwargs.get("json")) + raise RuntimeError("captured") + + with pytest.raises(Exception, match="captured"): + await litellm.aresponses( + model=DEEP_RESEARCH_MODEL, input="hi", stream=True, client=CapturingHandler() + ) + + assert captured["body"]["stream"] is True + + +class TestCoreModelStatelessSubset: + """Apodex core models reject anything that would persist state on their side.""" + + @pytest.mark.parametrize("model", [CORE_MODEL, CORE_MINI_MODEL]) + def test_store_is_pinned_false(self, model: str): + captured = _capture(model=model, input="hi") + assert captured["body"]["store"] is False + + def test_store_true_raises(self): + with pytest.raises(litellm.UnsupportedParamsError, match="store=True"): + _capture(model=CORE_MODEL, input="hi", store=True) + + def test_background_raises(self): + with pytest.raises(litellm.UnsupportedParamsError, match="background"): + _capture(model=CORE_MODEL, input="hi", background=True) + + def test_previous_response_id_raises(self): + with pytest.raises(litellm.UnsupportedParamsError, match="previous_response_id"): + _capture(model=CORE_MODEL, input="hi", previous_response_id="resp_1") + + def test_drop_params_strips_all_three(self, monkeypatch: pytest.MonkeyPatch): + monkeypatch.setattr(litellm, "drop_params", True) + captured = _capture( + model=CORE_MODEL, + input="hi", + store=True, + background=True, + previous_response_id="resp_1", + ) + + assert captured["body"]["store"] is False + assert "background" not in captured["body"] + assert "previous_response_id" not in captured["body"] + + def test_stateful_params_are_not_advertised(self): + supported = _responses_config("apodex-1.1").get_supported_openai_params("apodex-1.1") + assert "background" not in supported + assert "previous_response_id" not in supported + assert "max_output_tokens" in supported + + def test_max_output_tokens_still_passes_through(self): + captured = _capture(model=CORE_MODEL, input="hi", max_output_tokens=512) + assert captured["body"]["max_output_tokens"] == 512 + + +class TestDeepResearchKeepsState: + """The agent tiers survive client disconnects, so none of this may be stripped.""" + + def test_background_passes_through(self): + captured = _capture(model=DEEP_RESEARCH_MODEL, input="hi", background=True) + assert captured["body"]["background"] is True + + def test_store_and_previous_response_id_pass_through(self): + captured = _capture(model=DEEP_RESEARCH_MODEL, input="hi", store=True, previous_response_id="resp_1") + assert captured["body"]["store"] is True + assert captured["body"]["previous_response_id"] == "resp_1" + + def test_store_is_not_pinned_when_unset(self): + captured = _capture(model=DEEP_RESEARCH_MODEL, input="hi") + assert "store" not in captured["body"] + + def test_stateful_params_are_advertised(self): + supported = _responses_config("apodex-1-1-deep-research").get_supported_openai_params( + "apodex-1-1-deep-research" + ) + assert "background" in supported + assert "previous_response_id" in supported diff --git a/tests/test_litellm/llms/openai_like/test_apodex_provider.py b/tests/test_litellm/llms/openai_like/test_apodex_provider.py deleted file mode 100644 index cdcf9b4b50a..00000000000 --- a/tests/test_litellm/llms/openai_like/test_apodex_provider.py +++ /dev/null @@ -1,310 +0,0 @@ -""" -Tests for the Apodex provider (https://platform.apodex.ai/docs). -""" - -import json -from pathlib import Path - -import httpx -import openai -import pytest - -import litellm -from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler -from litellm.types.utils import LlmProviders -from litellm.utils import ProviderConfigManager - -REPO_ROOT = Path(__file__).parents[4] -CORE_MODEL = "apodex/apodex-1.1" -DEEP_RESEARCH_MODEL = "apodex/apodex-1-1-deep-research" - -CHAT_RESPONSE = { - "id": "chatcmpl-abc123", - "object": "chat.completion", - "created": 1712345678, - "model": "apodex-1.1", - "choices": [ - { - "index": 0, - "message": {"role": "assistant", "content": "ok", "reasoning_content": "let me think"}, - "finish_reason": "stop", - } - ], - "usage": { - "prompt_tokens": 1000, - "completion_tokens": 100, - "total_tokens": 1100, - "prompt_tokens_details": {"cached_tokens": 500}, - }, -} - -STREAM_BODY = ( - b'data: {"id":"chatcmpl-abc123","object":"chat.completion.chunk","created":1,"model":"apodex-1.1",' - b'"choices":[{"index":0,"delta":{"content":"ok"},"finish_reason":null}]}\n\n' - b"data: [DONE]\n\n" -) - - -@pytest.fixture(autouse=True) -def _apodex_env(monkeypatch: pytest.MonkeyPatch): - """Resolve models against the in-repo cost map, not the published one.""" - monkeypatch.setenv("APODEX_API_KEY", "sk-apodex-test") - monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") - monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) - yield - - -def _openai_client(captured: dict, *, stream: bool = False) -> openai.OpenAI: - def handler(request: httpx.Request) -> httpx.Response: - captured["url"] = str(request.url) - captured["body"] = json.loads(request.content) - if stream: - return httpx.Response(200, headers={"content-type": "text/event-stream"}, content=STREAM_BODY) - return httpx.Response(200, json=CHAT_RESPONSE) - - return openai.OpenAI( - api_key="sk-apodex-test", - base_url="https://api.apodex.ai/v1", - http_client=httpx.Client(transport=httpx.MockTransport(handler)), - ) - - -class TestApodexRegistration: - def test_provider_enum_and_lists(self): - assert LlmProviders.APODEX.value == "apodex" - assert "apodex" in litellm.provider_list - assert "apodex" in litellm.constants.openai_compatible_providers - - def test_json_provider_config(self): - from litellm.llms.openai_like.json_loader import JSONProviderRegistry - - apodex = JSONProviderRegistry.get("apodex") - assert apodex is not None - assert apodex.base_url == "https://api.apodex.ai/v1" - assert apodex.api_key_env == "APODEX_API_KEY" - assert apodex.api_base_env == "APODEX_API_BASE" - assert apodex.param_mappings["max_completion_tokens"] == "max_tokens" - assert apodex.supported_endpoints == ["/v1/chat/completions", "/v1/responses", "/v1/messages"] - assert JSONProviderRegistry.supports_responses_api("apodex") is True - - def test_provider_resolution(self): - model, provider, _, api_base = litellm.get_llm_provider(model=CORE_MODEL) - assert (model, provider, api_base) == ("apodex-1.1", "apodex", "https://api.apodex.ai/v1") - - def test_api_base_autodetection(self): - _, provider, api_key, _ = litellm.get_llm_provider(model="apodex-1.1", api_base="https://api.apodex.ai/v1") - assert provider == "apodex" - assert api_key == "sk-apodex-test" - - def test_explicit_api_base_and_key_win(self): - _, provider, api_key, api_base = litellm.get_llm_provider( - model=CORE_MODEL, api_base="https://gateway.internal/v1", api_key="sk-override" - ) - assert (provider, api_key, api_base) == ("apodex", "sk-override", "https://gateway.internal/v1") - - -class TestApodexStreamDefault: - """Apodex defaults `stream` to true, so a non-streaming call must pin it to false. - - Regression guard: the OpenAI SDK omits `stream` when it is false, which would make - litellm.completion() receive SSE and fail to parse it. - """ - - def test_chat_completion_pins_stream_false(self): - captured: dict = {} - response = litellm.completion( - model=CORE_MODEL, - messages=[{"role": "user", "content": "hi"}], - client=_openai_client(captured), - ) - - assert captured["url"] == "https://api.apodex.ai/v1/chat/completions" - assert captured["body"]["stream"] is False - assert captured["body"]["model"] == "apodex-1.1" - assert response.choices[0].message.reasoning_content == "let me think" - - def test_chat_completion_streaming_sends_stream_true(self): - captured: dict = {} - chunks = list( - litellm.completion( - model=CORE_MODEL, - messages=[{"role": "user", "content": "hi"}], - stream=True, - client=_openai_client(captured, stream=True), - ) - ) - - assert captured["body"]["stream"] is True - assert chunks - - def test_user_supplied_extra_body_is_preserved(self): - captured: dict = {} - litellm.completion( - model=CORE_MODEL, - messages=[{"role": "user", "content": "hi"}], - extra_body={"mcp_servers": [{"name": "docs", "url": "https://example.com/mcp"}]}, - client=_openai_client(captured), - ) - - assert captured["body"]["stream"] is False - assert captured["body"]["mcp_servers"] == [{"name": "docs", "url": "https://example.com/mcp"}] - - def test_responses_api_pins_stream_false(self): - captured: dict = {} - - class CapturingHandler(HTTPHandler): - def post(self, *args, **kwargs): - captured.update(url=kwargs.get("url"), body=kwargs.get("json")) - raise RuntimeError("captured") - - with pytest.raises(Exception): - litellm.responses(model=DEEP_RESEARCH_MODEL, input="hi", client=CapturingHandler()) - - assert captured["url"] == "https://api.apodex.ai/v1/responses" - assert captured["body"]["stream"] is False - assert captured["body"]["model"] == "apodex-1-1-deep-research" - - @pytest.mark.asyncio - async def test_responses_api_streaming_sends_stream_true(self): - captured: dict = {} - - class CapturingHandler(AsyncHTTPHandler): - async def post(self, *args, **kwargs): - captured.update(body=kwargs.get("json")) - raise RuntimeError("captured") - - with pytest.raises(Exception): - await litellm.aresponses(model=DEEP_RESEARCH_MODEL, input="hi", stream=True, client=CapturingHandler()) - - assert captured["body"]["stream"] is True - - def test_flag_is_opt_in_for_other_json_providers(self): - from litellm.llms.openai_like.json_loader import JSONProviderRegistry - - pinstripes = JSONProviderRegistry.get("pinstripes") - assert pinstripes is not None - assert "send_explicit_stream_false" not in pinstripes.special_handling - - config = ProviderConfigManager.get_provider_chat_config( - model="ps/glm-4.5-air", provider=LlmProviders.PINSTRIPES - ) - params = config.map_openai_params({}, {}, "ps/glm-4.5-air", False) - assert "stream" not in params - assert "stream" not in (params.get("extra_body") or {}) - - -class TestApodexToolSupport: - """Deep research tiers reject OpenAI-style tools; core models accept them.""" - - def test_deep_research_drops_tool_params(self): - config = ProviderConfigManager.get_provider_chat_config( - model="apodex-1-1-deep-research", provider=LlmProviders.APODEX - ) - supported = config.get_supported_openai_params("apodex-1-1-deep-research") - assert "tools" not in supported - assert "tool_choice" not in supported - - def test_core_model_keeps_tool_params(self): - config = ProviderConfigManager.get_provider_chat_config(model="apodex-1.1", provider=LlmProviders.APODEX) - supported = config.get_supported_openai_params("apodex-1.1") - assert "tools" in supported - assert "tool_choice" in supported - - def test_max_completion_tokens_maps_to_max_tokens(self): - config = ProviderConfigManager.get_provider_chat_config(model="apodex-1.1", provider=LlmProviders.APODEX) - params = config.map_openai_params({"max_completion_tokens": 512}, {}, "apodex-1.1", False) - assert params["max_tokens"] == 512 - assert "max_completion_tokens" not in params - - -class TestApodexAnthropicMessages: - """Apodex serves POST /v1/messages natively, so the payload is forwarded untranslated.""" - - def test_native_passthrough_config(self): - config = ProviderConfigManager.get_provider_anthropic_messages_config( - model="apodex-1.1", provider=LlmProviders.APODEX - ) - assert config is not None - assert type(config).__name__ == "JSONProviderAnthropicMessagesConfig" - assert ( - config.get_complete_url( - api_base=None, api_key=None, model="apodex-1.1", optional_params={}, litellm_params={} - ) - == "https://api.apodex.ai/v1/messages" - ) - - def test_headers_use_provider_api_key(self): - config = ProviderConfigManager.get_provider_anthropic_messages_config( - model="apodex-1.1", provider=LlmProviders.APODEX - ) - assert config is not None - headers, _ = config.validate_anthropic_messages_environment( - headers={}, model="apodex-1.1", messages=[], optional_params={}, litellm_params={} - ) - assert headers["authorization"] == "Bearer sk-apodex-test" - assert headers["anthropic-version"] == "2023-06-01" - - -class TestApodexModelMetadata: - @pytest.fixture(scope="class") - def model_cost(self) -> dict: - with open(REPO_ROOT / "model_prices_and_context_window.json") as f: - return json.load(f) - - def test_core_model_pricing(self, model_cost: dict): - info = model_cost["apodex/apodex-1.1"] - assert info["litellm_provider"] == "apodex" - assert info["mode"] == "chat" - assert info["max_input_tokens"] == 262144 - assert info["input_cost_per_token"] == 3e-07 - assert info["cache_read_input_token_cost"] == 3e-08 - assert info["output_cost_per_token"] == 3e-06 - # Requests over 200K input tokens are billed at 2x across every tier - assert info["input_cost_per_token_above_200k_tokens"] == 6e-07 - assert info["cache_read_input_token_cost_above_200k_tokens"] == 6e-08 - assert info["output_cost_per_token_above_200k_tokens"] == 6e-06 - assert info["supports_prompt_caching"] is True - assert info["supports_function_calling"] is True - assert info["supported_endpoints"] == ["/v1/chat/completions", "/v1/responses", "/v1/messages"] - - def test_deep_research_model_pricing(self, model_cost: dict): - info = model_cost["apodex/apodex-1-1-deep-research"] - assert info["max_input_tokens"] == 131072 - assert info["max_output_tokens"] == 65536 - assert info["input_cost_per_token"] == 5e-06 - assert info["output_cost_per_token"] == 2e-05 - assert info["supports_function_calling"] is False - assert info["supports_response_schema"] is False - assert info["supports_prompt_caching"] is False - assert info["supports_web_search"] is True - assert info["supported_endpoints"] == ["/v1/chat/completions", "/v1/responses"] - - def test_every_apodex_model_is_registered(self, model_cost: dict): - assert {key for key in model_cost if key.startswith("apodex/")} == { - "apodex/apodex-1.1", - "apodex/apodex-1.1-mini", - "apodex/apodex-1-1-deep-research", - "apodex/apodex-1-1-deep-solve", - "apodex/apodex-1-1-deep-discover", - "apodex/apodex-1-0-deep-research", - "apodex/apodex-1-0-deep-solve", - "apodex/apodex-1-0-deep-discover", - } - - def test_backup_cost_map_in_sync(self, model_cost: dict): - with open(REPO_ROOT / "litellm" / "model_prices_and_context_window_backup.json") as f: - backup = json.load(f) - for key in (key for key in model_cost if key.startswith("apodex/")): - assert backup[key] == model_cost[key], f"{key} differs between main and backup cost maps" - - def test_cost_tracks_cached_input_separately(self): - captured: dict = {} - response = litellm.completion( - model=CORE_MODEL, - messages=[{"role": "user", "content": "hi"}], - client=_openai_client(captured), - ) - - # 500 fresh input + 500 cached input + 100 output - expected = 500 * 3e-07 + 500 * 3e-08 + 100 * 3e-06 - assert litellm.completion_cost(response, model=CORE_MODEL) == pytest.approx(expected) diff --git a/tests/test_litellm/llms/openai_like/test_json_providers.py b/tests/test_litellm/llms/openai_like/test_json_providers.py index 442dd554885..c8743e1809d 100644 --- a/tests/test_litellm/llms/openai_like/test_json_providers.py +++ b/tests/test_litellm/llms/openai_like/test_json_providers.py @@ -245,63 +245,6 @@ class TestPinstripes: assert result["temperature"] == 0.7 -class TestTemperatureConstraints: - """`constraints` in providers.json clamp temperature before the request is sent.""" - - @staticmethod - def _config(constraints: dict): - from litellm.llms.openai_like.dynamic_config import create_config_class - from litellm.llms.openai_like.json_loader import SimpleProviderConfig - - provider = SimpleProviderConfig( - "constrained", - { - "base_url": "https://api.constrained.test/v1", - "api_key_env": "CONSTRAINED_API_KEY", - "constraints": constraints, - }, - ) - return create_config_class(provider)() - - def test_temperature_clamped_to_max(self): - config = self._config({"temperature_max": 1.0}) - result = config.map_openai_params({"temperature": 1.8}, {}, "some-model", False) - assert result["temperature"] == 1.0 - - def test_temperature_clamped_to_min(self): - config = self._config({"temperature_min": 0.1}) - result = config.map_openai_params({"temperature": 0.0}, {}, "some-model", False) - assert result["temperature"] == 0.1 - - def test_temperature_within_range_is_untouched(self): - config = self._config({"temperature_min": 0.1, "temperature_max": 1.0}) - result = config.map_openai_params({"temperature": 0.7}, {}, "some-model", False) - assert result["temperature"] == 0.7 - - def test_temperature_floor_applies_only_when_n_gt_1(self): - config = self._config({"temperature_min_with_n_gt_1": 0.3}) - - single = config.map_openai_params({"temperature": 0.0, "n": 1}, {}, "some-model", False) - assert single["temperature"] == 0.0 - - multiple = config.map_openai_params({"temperature": 0.0, "n": 2}, {}, "some-model", False) - assert multiple["temperature"] == 0.3 - - def test_no_constraints_leaves_temperature_alone(self): - config = self._config({}) - result = config.map_openai_params({"temperature": 1.9}, {}, "some-model", False) - assert result["temperature"] == 1.9 - - def test_caller_optional_params_are_not_mutated(self): - config = self._config({"temperature_max": 1.0}) - optional_params = {"temperature": 1.8} - result = config.map_openai_params({"max_tokens": 10}, optional_params, "some-model", False) - - assert result["temperature"] == 1.0 - assert result["max_tokens"] == 10 - assert optional_params == {"temperature": 1.8} - - class TestDarkbloom: def test_darkbloom_json_config_exists(self): from litellm.llms.openai_like.json_loader import JSONProviderRegistry diff --git a/type-discipline-budget.json b/type-discipline-budget.json index b6752ada511..8e55b1533ea 100644 --- a/type-discipline-budget.json +++ b/type-discipline-budget.json @@ -27,10 +27,10 @@ "limit": 0 }, "LIT010": { - "limit": 16707 + "limit": 16715 }, "LIT011": { - "limit": 5589 + "limit": 5593 }, "LIT012": { "limit": 4519 From eb42c9ea06b78bd6ed3f2a5da78a268957a3173b Mon Sep 17 00:00:00 2001 From: zhanghanduo Date: Sun, 16 Aug 2026 13:08:25 +0800 Subject: [PATCH 03/12] refactor(apodex): limit registration to 1.1 models --- ...odel_prices_and_context_window_backup.json | 69 ------------------- model_prices_and_context_window.json | 69 ------------------- .../llms/apodex/test_apodex_common_utils.py | 3 - .../test_apodex_messages_transformation.py | 3 - 4 files changed, 144 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 9d6a66e9327..6e9a4c90130 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -48417,75 +48417,6 @@ "supports_vision": false, "supports_web_search": true }, - "apodex/apodex-1-0-deep-research": { - "max_tokens": 16384, - "max_input_tokens": 262144, - "max_output_tokens": 16384, - "input_cost_per_token": 1e-05, - "output_cost_per_token": 4e-05, - "litellm_provider": "apodex", - "mode": "chat", - "source": "https://platform.apodex.ai/docs/pricing", - "supported_endpoints": [ - "/v1/chat/completions", - "/v1/responses" - ], - "supports_function_calling": false, - "supports_native_streaming": true, - "supports_prompt_caching": false, - "supports_reasoning": true, - "supports_response_schema": false, - "supports_system_messages": true, - "supports_tool_choice": false, - "supports_vision": false, - "supports_web_search": true - }, - "apodex/apodex-1-0-deep-solve": { - "max_tokens": 16384, - "max_input_tokens": 262144, - "max_output_tokens": 16384, - "input_cost_per_token": 1e-05, - "output_cost_per_token": 5e-05, - "litellm_provider": "apodex", - "mode": "chat", - "source": "https://platform.apodex.ai/docs/pricing", - "supported_endpoints": [ - "/v1/chat/completions", - "/v1/responses" - ], - "supports_function_calling": false, - "supports_native_streaming": true, - "supports_prompt_caching": false, - "supports_reasoning": true, - "supports_response_schema": false, - "supports_system_messages": true, - "supports_tool_choice": false, - "supports_vision": false, - "supports_web_search": true - }, - "apodex/apodex-1-0-deep-discover": { - "max_tokens": 262144, - "max_input_tokens": 131072, - "max_output_tokens": 262144, - "input_cost_per_token": 1e-05, - "output_cost_per_token": 0.0001, - "litellm_provider": "apodex", - "mode": "chat", - "source": "https://platform.apodex.ai/docs/pricing", - "supported_endpoints": [ - "/v1/chat/completions", - "/v1/responses" - ], - "supports_function_calling": false, - "supports_native_streaming": true, - "supports_prompt_caching": false, - "supports_reasoning": true, - "supports_response_schema": false, - "supports_system_messages": true, - "supports_tool_choice": false, - "supports_vision": false, - "supports_web_search": true - }, "fallback_generalizations": { "rules": [ { diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 9d6a66e9327..6e9a4c90130 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -48417,75 +48417,6 @@ "supports_vision": false, "supports_web_search": true }, - "apodex/apodex-1-0-deep-research": { - "max_tokens": 16384, - "max_input_tokens": 262144, - "max_output_tokens": 16384, - "input_cost_per_token": 1e-05, - "output_cost_per_token": 4e-05, - "litellm_provider": "apodex", - "mode": "chat", - "source": "https://platform.apodex.ai/docs/pricing", - "supported_endpoints": [ - "/v1/chat/completions", - "/v1/responses" - ], - "supports_function_calling": false, - "supports_native_streaming": true, - "supports_prompt_caching": false, - "supports_reasoning": true, - "supports_response_schema": false, - "supports_system_messages": true, - "supports_tool_choice": false, - "supports_vision": false, - "supports_web_search": true - }, - "apodex/apodex-1-0-deep-solve": { - "max_tokens": 16384, - "max_input_tokens": 262144, - "max_output_tokens": 16384, - "input_cost_per_token": 1e-05, - "output_cost_per_token": 5e-05, - "litellm_provider": "apodex", - "mode": "chat", - "source": "https://platform.apodex.ai/docs/pricing", - "supported_endpoints": [ - "/v1/chat/completions", - "/v1/responses" - ], - "supports_function_calling": false, - "supports_native_streaming": true, - "supports_prompt_caching": false, - "supports_reasoning": true, - "supports_response_schema": false, - "supports_system_messages": true, - "supports_tool_choice": false, - "supports_vision": false, - "supports_web_search": true - }, - "apodex/apodex-1-0-deep-discover": { - "max_tokens": 262144, - "max_input_tokens": 131072, - "max_output_tokens": 262144, - "input_cost_per_token": 1e-05, - "output_cost_per_token": 0.0001, - "litellm_provider": "apodex", - "mode": "chat", - "source": "https://platform.apodex.ai/docs/pricing", - "supported_endpoints": [ - "/v1/chat/completions", - "/v1/responses" - ], - "supports_function_calling": false, - "supports_native_streaming": true, - "supports_prompt_caching": false, - "supports_reasoning": true, - "supports_response_schema": false, - "supports_system_messages": true, - "supports_tool_choice": false, - "supports_vision": false, - "supports_web_search": true - }, "fallback_generalizations": { "rules": [ { diff --git a/tests/test_litellm/llms/apodex/test_apodex_common_utils.py b/tests/test_litellm/llms/apodex/test_apodex_common_utils.py index 1bb72ca13d6..bbbe1c2f052 100644 --- a/tests/test_litellm/llms/apodex/test_apodex_common_utils.py +++ b/tests/test_litellm/llms/apodex/test_apodex_common_utils.py @@ -23,9 +23,6 @@ DEEP_RESEARCH_MODELS = ( "apodex-1-1-deep-research", "apodex-1-1-deep-solve", "apodex-1-1-deep-discover", - "apodex-1-0-deep-research", - "apodex-1-0-deep-solve", - "apodex-1-0-deep-discover", ) diff --git a/tests/test_litellm/llms/apodex/test_apodex_messages_transformation.py b/tests/test_litellm/llms/apodex/test_apodex_messages_transformation.py index 8be810539e3..9d8c787c75b 100644 --- a/tests/test_litellm/llms/apodex/test_apodex_messages_transformation.py +++ b/tests/test_litellm/llms/apodex/test_apodex_messages_transformation.py @@ -18,9 +18,6 @@ DEEP_RESEARCH_MODELS = ( "apodex-1-1-deep-research", "apodex-1-1-deep-solve", "apodex-1-1-deep-discover", - "apodex-1-0-deep-research", - "apodex-1-0-deep-solve", - "apodex-1-0-deep-discover", ) From 4ace5c8db3cbaf6ba498d35326fec73cba3db4c1 Mon Sep 17 00:00:00 2001 From: zhanghanduo Date: Sun, 16 Aug 2026 13:37:54 +0800 Subject: [PATCH 04/12] fix(apodex): route deep research through responses --- .../messages/handler.py | 4 +- litellm/llms/apodex/chat/transformation.py | 4 +- .../llms/apodex/responses/transformation.py | 82 ++++++++++++++++++- .../test_apodex_messages_transformation.py | 11 ++- .../test_apodex_responses_transformation.py | 46 +++++++++++ 5 files changed, 137 insertions(+), 10 deletions(-) diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py index c4b5cc628e2..128f891386c 100644 --- a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py +++ b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py @@ -37,9 +37,9 @@ from ..utils import is_reasoning_auto_summary_enabled from .interceptors import get_messages_interceptors from .utils import AnthropicMessagesRequestUtils, mock_response -# Providers that are routed directly to the OpenAI Responses API instead of +# Providers that are routed directly to a Responses API instead of # going through chat/completions. -_RESPONSES_API_PROVIDERS: Final = frozenset({"openai"}) +_RESPONSES_API_PROVIDERS: Final = frozenset({"apodex", "openai"}) def _should_route_to_responses_api(custom_llm_provider: str | None) -> bool: diff --git a/litellm/llms/apodex/chat/transformation.py b/litellm/llms/apodex/chat/transformation.py index 300e3e57991..38f3757658b 100644 --- a/litellm/llms/apodex/chat/transformation.py +++ b/litellm/llms/apodex/chat/transformation.py @@ -109,9 +109,7 @@ class ApodexChatConfig(OpenAIGPTConfig): # request body by the SDK, so it survives that drop. requested_extra_body: Final = renamed.get("extra_body") extra_body: Final = ( - requested_extra_body - if isinstance(requested_extra_body, Mapping) - else {} # mutable-ok: JSON request body + requested_extra_body if isinstance(requested_extra_body, Mapping) else {} # mutable-ok: JSON request body ) return { # mutable-ok: JSON request body **renamed, diff --git a/litellm/llms/apodex/responses/transformation.py b/litellm/llms/apodex/responses/transformation.py index 04ca2e55c74..a7820757bc9 100644 --- a/litellm/llms/apodex/responses/transformation.py +++ b/litellm/llms/apodex/responses/transformation.py @@ -14,20 +14,34 @@ Ref: https://platform.apodex.ai/docs/responses-api https://platform.apodex.ai/docs/models """ +from __future__ import annotations + from collections.abc import Mapping -from typing import Final +from time import time +from typing import TYPE_CHECKING, Final + +import httpx +from pydantic import TypeAdapter import litellm from litellm.llms.openai.responses.transformation import OpenAIResponsesAPIConfig -from litellm.types.llms.openai import ResponsesAPIOptionalRequestParams +from litellm.types.llms.openai import ( + ResponsesAPIOptionalRequestParams, + ResponsesAPIResponse, + ResponsesAPIStreamingResponse, +) from litellm.types.router import GenericLiteLLMParams from litellm.types.utils import LlmProviders -from ..common_utils import get_apodex_api_key, is_deep_research_model +from ..common_utils import get_apodex_api_base, get_apodex_api_key, is_deep_research_model + +if TYPE_CHECKING: + from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj # Rejected by the core models with HTTP 400: there is no server-side conversation # to resume and requests are always executed inline. _STATEFUL_PARAMS: Final = ("previous_response_id", "background") +_CANCEL_RESPONSE_ADAPTER: Final = TypeAdapter(dict[str, object]) class ApodexResponsesConfig(OpenAIResponsesAPIConfig): @@ -53,6 +67,68 @@ class ApodexResponsesConfig(OpenAIResponsesAPIConfig): "Authorization": f"Bearer {api_key}", } + def get_complete_url( + self, + api_base: str | None, + litellm_params: dict, # mutable-ok: matches the base-class signature + ) -> str: + resolved_base: Final = get_apodex_api_base(api_base).rstrip("/") + return f"{resolved_base}/responses" + + def transform_cancel_response_api_response( + self, + raw_response: httpx.Response, + logging_obj: LiteLLMLoggingObj, + ) -> ResponsesAPIResponse: + payload: Final = _CANCEL_RESPONSE_ADAPTER.validate_json(raw_response.content) + normalized_response: Final = httpx.Response( + status_code=raw_response.status_code, + headers=raw_response.headers, + json={ + **payload, + "created_at": payload.get("created_at", int(time())), + "output": payload.get("output", []), + }, + ) + return super().transform_cancel_response_api_response( + raw_response=normalized_response, + logging_obj=logging_obj, + ) + + def transform_streaming_response( + self, + model: str, + parsed_chunk: dict, # mutable-ok: matches the base-class signature + logging_obj: LiteLLMLoggingObj, + ) -> ResponsesAPIStreamingResponse: + swarm: Final = parsed_chunk.get("swarm") + swarm_data: Final = swarm.get("data") if isinstance(swarm, dict) else None + if ( + parsed_chunk.get("type") != "response.swarm.llm_delta" + or not isinstance(swarm_data, dict) + or swarm_data.get("channel") != "output_text" + or not isinstance(swarm_data.get("delta"), str) + ): + return super().transform_streaming_response( + model=model, + parsed_chunk=parsed_chunk, + logging_obj=logging_obj, + ) + + response_id: Final = str(parsed_chunk.get("response_id", "")) + return super().transform_streaming_response( + model=model, + parsed_chunk={ + "type": "response.output_text.delta", + "item_id": f"msg_{response_id}", + "output_index": 0, + "content_index": 0, + "delta": swarm_data["delta"], + "sequence_number": parsed_chunk.get("sequence_number", 0), + }, + logging_obj=logging_obj, + ) + def get_supported_openai_params(self, model: str) -> list: # mutable-ok: matches the base-class signature inherited: Final = super().get_supported_openai_params(model) if is_deep_research_model(model): diff --git a/tests/test_litellm/llms/apodex/test_apodex_messages_transformation.py b/tests/test_litellm/llms/apodex/test_apodex_messages_transformation.py index 9d8c787c75b..09427b31a2f 100644 --- a/tests/test_litellm/llms/apodex/test_apodex_messages_transformation.py +++ b/tests/test_litellm/llms/apodex/test_apodex_messages_transformation.py @@ -58,10 +58,17 @@ class TestNativePassthroughRouting: @pytest.mark.parametrize("model", DEEP_RESEARCH_MODELS) def test_deep_research_models_fall_back_to_translation(self, model: str): - """No native config means LiteLLM translates to chat completions, which works, - instead of forwarding to a path Apodex does not serve for these tiers.""" + """No native config means LiteLLM uses a protocol translation instead of + forwarding to a path Apodex does not serve for these tiers.""" assert _messages_config(model) is None + def test_deep_research_translation_uses_responses_api(self): + from litellm.llms.anthropic.experimental_pass_through.messages.handler import ( + _should_route_to_responses_api, + ) + + assert _should_route_to_responses_api("apodex") is True + class TestNativePassthroughRequest: def test_url_targets_the_native_messages_path(self): diff --git a/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py b/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py index d343f4553c8..44eb0cb4b1e 100644 --- a/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py +++ b/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py @@ -6,6 +6,7 @@ Research tiers keep server-side state, so the parameter contract is keyed off the model rather than applied provider-wide. """ +import httpx import pytest import litellm @@ -80,6 +81,17 @@ class TestConfigSelection: def test_request_targets_the_apodex_responses_url(self): assert _capture(model=CORE_MODEL, input="hi")["url"] == "https://api.apodex.ai/v1/responses" + def test_polling_without_model_resolution_targets_apodex(self): + config = _responses_config("apodex-1-1-deep-research") + assert config.get_complete_url(api_base=None, litellm_params={}) == "https://api.apodex.ai/v1/responses" + + def test_polling_honours_an_explicit_api_base(self): + config = _responses_config("apodex-1-1-deep-research") + assert ( + config.get_complete_url(api_base="https://gateway.apodex.test/v1/", litellm_params={}) + == "https://gateway.apodex.test/v1/responses" + ) + def test_request_honours_an_api_base_override(self, monkeypatch: pytest.MonkeyPatch): monkeypatch.setenv("APODEX_API_BASE", "https://env.apodex.test/v1") assert _capture(model=CORE_MODEL, input="hi")["url"] == "https://env.apodex.test/v1/responses" @@ -175,3 +187,37 @@ class TestDeepResearchKeepsState: ) assert "background" in supported assert "previous_response_id" in supported + + def test_minimal_cancel_response_is_normalized(self): + config = _responses_config("apodex-1-1-deep-research") + response = config.transform_cancel_response_api_response( + raw_response=httpx.Response( + 200, + json={"id": "resp_1", "object": "response", "status": "cancelled"}, + ), + logging_obj=None, + ) + + assert response.id == "resp_1" + assert response.status == "cancelled" + assert response.output == [] + assert response.created_at > 0 + + def test_deep_research_output_delta_is_normalized(self): + config = _responses_config("apodex-1-1-deep-research") + event = config.transform_streaming_response( + model="apodex-1-1-deep-research", + parsed_chunk={ + "type": "response.swarm.llm_delta", + "response_id": "w_123", + "sequence_number": 12, + "swarm": { + "agent_id": "reporter", + "data": {"channel": "output_text", "delta": "final answer"}, + }, + }, + logging_obj=None, + ) + assert event.type == "response.output_text.delta" + assert event.item_id == "msg_w_123" + assert event.delta == "final answer" From c3f8e5aa6722115cad91820c590e891054481b8c Mon Sep 17 00:00:00 2001 From: zhanghanduo Date: Sun, 16 Aug 2026 17:00:22 +0800 Subject: [PATCH 05/12] fix(apodex): correct model metadata and the cancel-response rebuild Cross-checked the provider against platform.apodex.ai/docs and a live GET /v1/models call. - apodex-1.1 and apodex-1.1-mini advertised 256K max output; /v1/models reports 65536, and max_tokens is the legacy alias of max_output_tokens - apodex-1-1-deep-discover is Responses-API-only; /v1/chat/completions answers 400 unsupported_api for the Discover tiers - core models do not support response_format, so state it explicitly - transform_cancel_response_api_response carried Content-Encoding over to a response whose body it had already replaced, so httpx tried to decompress plain JSON on read. A non-JSON body (the gateway answers a timed-out cancel with an HTML 504) also escaped as a pydantic ValidationError instead of the provider error - drop the undocumented response.swarm.llm_delta mapping - only the Deep Research tiers default stream to true; the core models follow OpenAI and default it to false --- .../get_llm_provider_logic.py | 1 - .../messages/handler.py | 3 +- litellm/llms/apodex/chat/transformation.py | 13 +++-- .../llms/apodex/responses/transformation.py | 57 ++++++------------- ...odel_prices_and_context_window_backup.json | 11 ++-- model_prices_and_context_window.json | 11 ++-- .../apodex/test_apodex_chat_transformation.py | 7 ++- .../llms/apodex/test_apodex_common_utils.py | 14 +++++ .../test_apodex_responses_transformation.py | 54 ++++++++++++------ 9 files changed, 94 insertions(+), 77 deletions(-) diff --git a/litellm/litellm_core_utils/get_llm_provider_logic.py b/litellm/litellm_core_utils/get_llm_provider_logic.py index ac1faeda645..4ea544b5baa 100644 --- a/litellm/litellm_core_utils/get_llm_provider_logic.py +++ b/litellm/litellm_core_utils/get_llm_provider_logic.py @@ -351,7 +351,6 @@ def get_llm_provider( dynamic_api_key = get_secret_str("META_API_KEY") elif endpoint == litellm.ApodexChatConfig.API_BASE_URL: custom_llm_provider = "apodex" # rebind-ok: dispatch chain resolves in place - # rebind-ok: dispatch chain resolves in place dynamic_api_key = litellm.ApodexChatConfig.get_api_key() if api_base is not None and not isinstance(api_base, str): diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py index 128f891386c..bafab2c9244 100644 --- a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py +++ b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py @@ -529,7 +529,8 @@ def anthropic_messages_handler( anthropic_messages_provider_config = OpenAILikeAnthropicMessagesConfig() if anthropic_messages_provider_config is None: - # Route to Responses API for OpenAI / Azure, chat/completions for everything else. + # Route to a Responses API for the providers that serve one, chat/completions + # for everything else. _shared_kwargs: Final = dict( max_tokens=max_tokens, messages=messages, diff --git a/litellm/llms/apodex/chat/transformation.py b/litellm/llms/apodex/chat/transformation.py index 38f3757658b..4db9bc9fac1 100644 --- a/litellm/llms/apodex/chat/transformation.py +++ b/litellm/llms/apodex/chat/transformation.py @@ -1,12 +1,14 @@ """ Apodex chat completions — OpenAI-compatible, with two provider quirks: -- `stream` defaults to true upstream, so a non-streaming call has to say so - explicitly or Apodex answers with SSE that a plain call cannot parse +- the Deep Research tiers default `stream` to true, so a non-streaming call has + to say so explicitly or Apodex answers with SSE that a plain call cannot + parse. The core models follow OpenAI and default it to false - the Deep Research tiers ignore sampling parameters and reject OpenAI-style tools; only the core models take them Ref: https://platform.apodex.ai/docs/chat-completions + https://platform.apodex.ai/docs/models """ from collections.abc import Mapping @@ -104,9 +106,10 @@ class ApodexChatConfig(OpenAIGPTConfig): if renamed.get("stream"): return renamed - # The OpenAI SDK drops `stream` from the body when it is false, which would - # leave Apodex on its streaming default. extra_body is merged into the - # request body by the SDK, so it survives that drop. + # The OpenAI chat handler pops `stream` out of the params it forwards + # (litellm/llms/openai/openai.py), which would leave the Deep Research tiers + # on their streaming default. extra_body is merged into the request body + # further down, so it survives that pop. requested_extra_body: Final = renamed.get("extra_body") extra_body: Final = ( requested_extra_body if isinstance(requested_extra_body, Mapping) else {} # mutable-ok: JSON request body diff --git a/litellm/llms/apodex/responses/transformation.py b/litellm/llms/apodex/responses/transformation.py index a7820757bc9..e8014905846 100644 --- a/litellm/llms/apodex/responses/transformation.py +++ b/litellm/llms/apodex/responses/transformation.py @@ -8,7 +8,8 @@ provider-wide: - core models are a stateless subset: `store` is forced to false, and `previous_response_id` or `background` come back as HTTP 400 - the Deep Research tiers keep server-side state, so they take all three -- both default `stream` to true, so a non-streaming call has to say so +- the Deep Research tiers default `stream` to true, so a non-streaming call has + to say so; pinning it for the core models too keeps one code path Ref: https://platform.apodex.ai/docs/responses-api https://platform.apodex.ai/docs/models @@ -21,14 +22,13 @@ from time import time from typing import TYPE_CHECKING, Final import httpx -from pydantic import TypeAdapter +from pydantic import TypeAdapter, ValidationError import litellm from litellm.llms.openai.responses.transformation import OpenAIResponsesAPIConfig from litellm.types.llms.openai import ( ResponsesAPIOptionalRequestParams, ResponsesAPIResponse, - ResponsesAPIStreamingResponse, ) from litellm.types.router import GenericLiteLLMParams from litellm.types.utils import LlmProviders @@ -42,6 +42,7 @@ if TYPE_CHECKING: # to resume and requests are always executed inline. _STATEFUL_PARAMS: Final = ("previous_response_id", "background") _CANCEL_RESPONSE_ADAPTER: Final = TypeAdapter(dict[str, object]) +_BODY_FRAMING_HEADERS: Final = frozenset({"content-encoding", "content-length"}) class ApodexResponsesConfig(OpenAIResponsesAPIConfig): @@ -80,10 +81,22 @@ class ApodexResponsesConfig(OpenAIResponsesAPIConfig): raw_response: httpx.Response, logging_obj: LiteLLMLoggingObj, ) -> ResponsesAPIResponse: - payload: Final = _CANCEL_RESPONSE_ADAPTER.validate_json(raw_response.content) + """Backfill the fields Apodex omits from a cancel payload but ResponsesAPIResponse requires.""" + try: + payload: Final = _CANCEL_RESPONSE_ADAPTER.validate_json(raw_response.content) + except ValidationError: + return super().transform_cancel_response_api_response( + raw_response=raw_response, + logging_obj=logging_obj, + ) + normalized_response: Final = httpx.Response( status_code=raw_response.status_code, - headers=raw_response.headers, + # Content-Encoding and Content-Length describe the body being replaced here; + # carrying them over makes httpx try to decompress plain JSON on read. + headers={ + name: value for name, value in raw_response.headers.items() if name.lower() not in _BODY_FRAMING_HEADERS + }, json={ **payload, "created_at": payload.get("created_at", int(time())), @@ -95,40 +108,6 @@ class ApodexResponsesConfig(OpenAIResponsesAPIConfig): logging_obj=logging_obj, ) - def transform_streaming_response( - self, - model: str, - parsed_chunk: dict, # mutable-ok: matches the base-class signature - logging_obj: LiteLLMLoggingObj, - ) -> ResponsesAPIStreamingResponse: - swarm: Final = parsed_chunk.get("swarm") - swarm_data: Final = swarm.get("data") if isinstance(swarm, dict) else None - if ( - parsed_chunk.get("type") != "response.swarm.llm_delta" - or not isinstance(swarm_data, dict) - or swarm_data.get("channel") != "output_text" - or not isinstance(swarm_data.get("delta"), str) - ): - return super().transform_streaming_response( - model=model, - parsed_chunk=parsed_chunk, - logging_obj=logging_obj, - ) - - response_id: Final = str(parsed_chunk.get("response_id", "")) - return super().transform_streaming_response( - model=model, - parsed_chunk={ - "type": "response.output_text.delta", - "item_id": f"msg_{response_id}", - "output_index": 0, - "content_index": 0, - "delta": swarm_data["delta"], - "sequence_number": parsed_chunk.get("sequence_number", 0), - }, - logging_obj=logging_obj, - ) - def get_supported_openai_params(self, model: str) -> list: # mutable-ok: matches the base-class signature inherited: Final = super().get_supported_openai_params(model) if is_deep_research_model(model): diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 6e9a4c90130..250477bd81d 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -48297,9 +48297,9 @@ "supports_audio_output": true }, "apodex/apodex-1.1": { - "max_tokens": 262144, + "max_tokens": 65536, "max_input_tokens": 262144, - "max_output_tokens": 262144, + "max_output_tokens": 65536, "input_cost_per_token": 3e-07, "cache_read_input_token_cost": 3e-08, "output_cost_per_token": 3e-06, @@ -48318,14 +48318,15 @@ "supports_native_streaming": true, "supports_prompt_caching": true, "supports_reasoning": true, + "supports_response_schema": false, "supports_system_messages": true, "supports_tool_choice": true, "supports_vision": false }, "apodex/apodex-1.1-mini": { - "max_tokens": 262144, + "max_tokens": 65536, "max_input_tokens": 262144, - "max_output_tokens": 262144, + "max_output_tokens": 65536, "input_cost_per_token": 1e-07, "cache_read_input_token_cost": 1e-08, "output_cost_per_token": 1e-06, @@ -48344,6 +48345,7 @@ "supports_native_streaming": true, "supports_prompt_caching": true, "supports_reasoning": true, + "supports_response_schema": false, "supports_system_messages": true, "supports_tool_choice": true, "supports_vision": false @@ -48404,7 +48406,6 @@ "mode": "chat", "source": "https://platform.apodex.ai/docs/pricing", "supported_endpoints": [ - "/v1/chat/completions", "/v1/responses" ], "supports_function_calling": false, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 6e9a4c90130..250477bd81d 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -48297,9 +48297,9 @@ "supports_audio_output": true }, "apodex/apodex-1.1": { - "max_tokens": 262144, + "max_tokens": 65536, "max_input_tokens": 262144, - "max_output_tokens": 262144, + "max_output_tokens": 65536, "input_cost_per_token": 3e-07, "cache_read_input_token_cost": 3e-08, "output_cost_per_token": 3e-06, @@ -48318,14 +48318,15 @@ "supports_native_streaming": true, "supports_prompt_caching": true, "supports_reasoning": true, + "supports_response_schema": false, "supports_system_messages": true, "supports_tool_choice": true, "supports_vision": false }, "apodex/apodex-1.1-mini": { - "max_tokens": 262144, + "max_tokens": 65536, "max_input_tokens": 262144, - "max_output_tokens": 262144, + "max_output_tokens": 65536, "input_cost_per_token": 1e-07, "cache_read_input_token_cost": 1e-08, "output_cost_per_token": 1e-06, @@ -48344,6 +48345,7 @@ "supports_native_streaming": true, "supports_prompt_caching": true, "supports_reasoning": true, + "supports_response_schema": false, "supports_system_messages": true, "supports_tool_choice": true, "supports_vision": false @@ -48404,7 +48406,6 @@ "mode": "chat", "source": "https://platform.apodex.ai/docs/pricing", "supported_endpoints": [ - "/v1/chat/completions", "/v1/responses" ], "supports_function_calling": false, diff --git a/tests/test_litellm/llms/apodex/test_apodex_chat_transformation.py b/tests/test_litellm/llms/apodex/test_apodex_chat_transformation.py index 61fccf2483a..da31464d5cb 100644 --- a/tests/test_litellm/llms/apodex/test_apodex_chat_transformation.py +++ b/tests/test_litellm/llms/apodex/test_apodex_chat_transformation.py @@ -99,10 +99,11 @@ class TestProviderResolution: class TestStreamDefault: - """Apodex defaults `stream` to true, so a non-streaming call has to pin it to false. + """The Deep Research tiers default `stream` to true, so a non-streaming call must pin it false. - Regression guard: the OpenAI SDK drops `stream` from the body when it is false, - which would leave Apodex streaming SSE at a call that cannot parse it. + Regression guard: the OpenAI chat handler pops `stream` out of the params it + forwards, which would leave those tiers streaming SSE at a call that cannot + parse it. The core models default to false but are pinned the same way. """ def test_non_streaming_call_pins_stream_false(self): diff --git a/tests/test_litellm/llms/apodex/test_apodex_common_utils.py b/tests/test_litellm/llms/apodex/test_apodex_common_utils.py index bbbe1c2f052..2da109c6ec4 100644 --- a/tests/test_litellm/llms/apodex/test_apodex_common_utils.py +++ b/tests/test_litellm/llms/apodex/test_apodex_common_utils.py @@ -99,6 +99,9 @@ class TestModelMetadata: assert info["litellm_provider"] == "apodex" assert info["mode"] == "chat" assert info["max_input_tokens"] == 262144 + # GET /v1/models reports max_completion_tokens 65536, well under the context window + assert info["max_output_tokens"] == 65536 + assert info["max_tokens"] == info["max_output_tokens"] assert info["input_cost_per_token"] == 3e-07 assert info["cache_read_input_token_cost"] == 3e-08 assert info["output_cost_per_token"] == 3e-06 @@ -126,6 +129,17 @@ class TestModelMetadata: """Apodex serves /v1/messages for the core models only.""" assert "/v1/messages" not in model_cost[f"apodex/{model}"]["supported_endpoints"] + def test_discover_is_responses_only(self, model_cost: dict): + """The Discover tiers answer 400 unsupported_api on /v1/chat/completions.""" + assert model_cost["apodex/apodex-1-1-deep-discover"]["supported_endpoints"] == ["/v1/responses"] + + @pytest.mark.parametrize("model", ("apodex-1-1-deep-research", "apodex-1-1-deep-solve")) + def test_the_other_deep_tiers_keep_chat_completions(self, model_cost: dict, model: str): + assert model_cost[f"apodex/{model}"]["supported_endpoints"] == [ + "/v1/chat/completions", + "/v1/responses", + ] + def test_backup_cost_map_in_sync(self, model_cost: dict): with open(REPO_ROOT / "litellm" / "model_prices_and_context_window_backup.json") as f: backup = json.load(f) diff --git a/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py b/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py index 44eb0cb4b1e..ac8e6db7cc4 100644 --- a/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py +++ b/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py @@ -6,6 +6,8 @@ Research tiers keep server-side state, so the parameter contract is keyed off the model rather than applied provider-wide. """ +import gzip + import httpx import pytest @@ -113,9 +115,7 @@ class TestStreamDefault: raise RuntimeError("captured") with pytest.raises(Exception, match="captured"): - await litellm.aresponses( - model=DEEP_RESEARCH_MODEL, input="hi", stream=True, client=CapturingHandler() - ) + await litellm.aresponses(model=DEEP_RESEARCH_MODEL, input="hi", stream=True, client=CapturingHandler()) assert captured["body"]["stream"] is True @@ -203,21 +203,39 @@ class TestDeepResearchKeepsState: assert response.output == [] assert response.created_at > 0 - def test_deep_research_output_delta_is_normalized(self): - config = _responses_config("apodex-1-1-deep-research") - event = config.transform_streaming_response( - model="apodex-1-1-deep-research", - parsed_chunk={ - "type": "response.swarm.llm_delta", - "response_id": "w_123", - "sequence_number": 12, - "swarm": { - "agent_id": "reporter", - "data": {"channel": "output_text", "delta": "final answer"}, - }, + def test_cancel_response_survives_a_compressed_upstream_response(self): + """The body is rebuilt, so the original framing headers must not follow it. + + httpx decodes on read, so carrying Content-Encoding over from the compressed + upstream response makes it try to gunzip the plain JSON replacement. + """ + body = b'{"id": "resp_1", "object": "response", "status": "cancelled"}' + # As it arrives off the wire: httpx decodes the body but leaves the header in place + upstream = httpx.Response( + 200, + headers={ + "content-encoding": "gzip", + "x-ratelimit-remaining-requests": "42", }, + content=gzip.compress(body), + ) + assert upstream.content == body + + config = _responses_config("apodex-1-1-deep-research") + response = config.transform_cancel_response_api_response( + raw_response=upstream, logging_obj=None, ) - assert event.type == "response.output_text.delta" - assert event.item_id == "msg_w_123" - assert event.delta == "final answer" + + assert response.id == "resp_1" + assert response.status == "cancelled" + assert response._hidden_params["headers"]["x-ratelimit-remaining-requests"] == "42" + + def test_non_json_cancel_body_raises_the_provider_error(self): + """The gateway answers a timed-out cancel with an HTML 504, not the JSON envelope.""" + config = _responses_config("apodex-1-1-deep-research") + with pytest.raises(Exception, match="gateway timeout"): + config.transform_cancel_response_api_response( + raw_response=httpx.Response(504, content=b"gateway timeout"), + logging_obj=None, + ) From 64dcb95268e066d39078ced66a01f32ba49da8b6 Mon Sep 17 00:00:00 2001 From: zhanghanduo Date: Sun, 16 Aug 2026 17:42:46 +0800 Subject: [PATCH 06/12] Revert "drop the undocumented response.swarm.llm_delta mapping" A live stream against apodex-1-1-deep-research shows the event is real and load-bearing: 181 of the 193 events are response.swarm.llm_delta and response.output_text.delta never appears, so without the mapping the answer text only arrives in the final response.completed snapshot. Restores the transform with the provenance recorded in a docstring, and covers it with the payload shape captured off the wire, including the reasoning channel that carries 176 of those deltas and must not be mistaken for the answer. --- .../llms/apodex/responses/transformation.py | 43 ++++++++++++ .../test_apodex_responses_transformation.py | 65 +++++++++++++++++++ 2 files changed, 108 insertions(+) diff --git a/litellm/llms/apodex/responses/transformation.py b/litellm/llms/apodex/responses/transformation.py index e8014905846..065b7a475ec 100644 --- a/litellm/llms/apodex/responses/transformation.py +++ b/litellm/llms/apodex/responses/transformation.py @@ -29,6 +29,7 @@ from litellm.llms.openai.responses.transformation import OpenAIResponsesAPIConfi from litellm.types.llms.openai import ( ResponsesAPIOptionalRequestParams, ResponsesAPIResponse, + ResponsesAPIStreamingResponse, ) from litellm.types.router import GenericLiteLLMParams from litellm.types.utils import LlmProviders @@ -108,6 +109,48 @@ class ApodexResponsesConfig(OpenAIResponsesAPIConfig): logging_obj=logging_obj, ) + def transform_streaming_response( + self, + model: str, + parsed_chunk: dict, # mutable-ok: matches the base-class signature + logging_obj: LiteLLMLoggingObj, + ) -> ResponsesAPIStreamingResponse: + """Surface the Deep Research answer text as the OpenAI delta event callers expect. + + Observed live, not documented: a Deep Research stream carries its text in + `response.swarm.llm_delta` and never emits `response.output_text.delta`, so + without this the answer arrives only in the final `response.completed` + snapshot. `channel` splits the agent's reasoning from its answer; everything + else falls through to the base class as a GenericEvent. + """ + swarm: Final = parsed_chunk.get("swarm") + swarm_data: Final = swarm.get("data") if isinstance(swarm, dict) else None + if ( + parsed_chunk.get("type") != "response.swarm.llm_delta" + or not isinstance(swarm_data, dict) + or swarm_data.get("channel") != "output_text" + or not isinstance(swarm_data.get("delta"), str) + ): + return super().transform_streaming_response( + model=model, + parsed_chunk=parsed_chunk, + logging_obj=logging_obj, + ) + + response_id: Final = str(parsed_chunk.get("response_id", "")) + return super().transform_streaming_response( + model=model, + parsed_chunk={ # mutable-ok: JSON event payload + "type": "response.output_text.delta", + "item_id": f"msg_{response_id}", + "output_index": 0, + "content_index": 0, + "delta": swarm_data["delta"], + "sequence_number": parsed_chunk.get("sequence_number", 0), + }, + logging_obj=logging_obj, + ) + def get_supported_openai_params(self, model: str) -> list: # mutable-ok: matches the base-class signature inherited: Final = super().get_supported_openai_params(model) if is_deep_research_model(model): diff --git a/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py b/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py index ac8e6db7cc4..789911126c3 100644 --- a/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py +++ b/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py @@ -231,6 +231,71 @@ class TestDeepResearchKeepsState: assert response.status == "cancelled" assert response._hidden_params["headers"]["x-ratelimit-remaining-requests"] == "42" + def test_output_text_delta_is_normalized(self): + """A Deep Research stream never emits response.output_text.delta of its own. + + Payload shape captured from a live stream; the extra swarm.data keys ride + along untouched and must not affect the mapping. + """ + config = _responses_config("apodex-1-1-deep-research") + event = config.transform_streaming_response( + model="apodex-1-1-deep-research", + parsed_chunk={ + "type": "response.swarm.llm_delta", + "created_at": 1786873219.3543231, + "response_id": "w_c4b77c96", + "sequence_number": 12, + "swarm": { + "agent_id": "stateful_react", + "data": { + "channel": "output_text", + "delta": "Hello there, friend!", + "delta_index": 0, + "call_id": "llm_fce8e965", + "turn": 1, + }, + }, + }, + logging_obj=None, + ) + + assert event.type == "response.output_text.delta" + assert event.item_id == "msg_w_c4b77c96" + assert event.delta == "Hello there, friend!" + assert event.sequence_number == 12 + + @pytest.mark.parametrize( + "channel", + ("reasoning", None), + ids=("reasoning-channel", "no-channel"), + ) + def test_non_answer_deltas_are_not_claimed_as_output_text(self, channel): + """Most of the stream is the agent thinking; only `output_text` is the answer.""" + data = {"delta": "The"} if channel is None else {"channel": channel, "delta": "The"} + config = _responses_config("apodex-1-1-deep-research") + event = config.transform_streaming_response( + model="apodex-1-1-deep-research", + parsed_chunk={ + "type": "response.swarm.llm_delta", + "response_id": "w_c4b77c96", + "sequence_number": 7, + "swarm": {"agent_id": "stateful_react", "data": data}, + }, + logging_obj=None, + ) + + assert event.type == "response.swarm.llm_delta" + + def test_documented_events_pass_through_untouched(self): + config = _responses_config("apodex-1-1-deep-research") + event = config.transform_streaming_response( + model="apodex-1-1-deep-research", + parsed_chunk={"type": "response.in_progress", "sequence_number": 2}, + logging_obj=None, + ) + + assert event.type == "response.in_progress" + def test_non_json_cancel_body_raises_the_provider_error(self): """The gateway answers a timed-out cancel with an HTML 504, not the JSON envelope.""" config = _responses_config("apodex-1-1-deep-research") From 190a3e01ecd96382520b24768e04f34ca18244ae Mon Sep 17 00:00:00 2001 From: zhanghanduo Date: Sun, 16 Aug 2026 18:01:27 +0800 Subject: [PATCH 07/12] fix(apodex): harden provider routing metadata --- litellm/llms/apodex/chat/transformation.py | 35 ++++++++++++++++++- .../provider_endpoints_support_backup.json | 17 +++++++++ .../apodex/test_apodex_chat_transformation.py | 11 ++++++ .../llms/apodex/test_apodex_common_utils.py | 6 ++++ .../test_apodex_messages_transformation.py | 35 ++++++++++++++++--- 5 files changed, 98 insertions(+), 6 deletions(-) diff --git a/litellm/llms/apodex/chat/transformation.py b/litellm/llms/apodex/chat/transformation.py index 4db9bc9fac1..a1b13633f66 100644 --- a/litellm/llms/apodex/chat/transformation.py +++ b/litellm/llms/apodex/chat/transformation.py @@ -15,6 +15,7 @@ from collections.abc import Mapping from typing import Final from litellm.llms.openai.chat.gpt_transformation import OpenAIGPTConfig +from litellm.types.llms.openai import AllMessageValues from ..common_utils import ( APODEX_API_BASE_URL, @@ -46,6 +47,8 @@ _CORE_PARAMS: Final = ( "parallel_tool_calls", ) +_PIN_NON_STREAMING: Final = "_apodex_pin_non_streaming" + class ApodexChatConfig(OpenAIGPTConfig): """ @@ -116,5 +119,35 @@ class ApodexChatConfig(OpenAIGPTConfig): ) return { # mutable-ok: JSON request body **renamed, - "extra_body": {"stream": False, **extra_body}, # mutable-ok: JSON request body + _PIN_NON_STREAMING: True, + "extra_body": {**extra_body, "stream": False}, # mutable-ok: JSON request body + } + + def transform_request( + self, + model: str, + messages: list[AllMessageValues], + optional_params: dict, # mutable-ok: matches the base-class signature + litellm_params: dict, # mutable-ok: matches the base-class signature + headers: dict, # mutable-ok: matches the base-class signature + ) -> dict: # mutable-ok: JSON request body + """Apply the non-streaming pin after LiteLLM merges caller ``extra_body``.""" + pin_non_streaming: Final = bool(optional_params.pop(_PIN_NON_STREAMING, False)) + transformed: Final = super().transform_request( + model=model, + messages=messages, + optional_params=optional_params, + litellm_params=litellm_params, + headers=headers, + ) + if not pin_non_streaming: + return transformed + + requested_extra_body: Final = transformed.get("extra_body") + extra_body: Final = ( + requested_extra_body if isinstance(requested_extra_body, Mapping) else {} # mutable-ok: JSON request body + ) + return { # mutable-ok: JSON request body + **transformed, + "extra_body": {**extra_body, "stream": False}, # mutable-ok: JSON request body } diff --git a/litellm/provider_endpoints_support_backup.json b/litellm/provider_endpoints_support_backup.json index dd7712aabca..f15dcb2dfb2 100644 --- a/litellm/provider_endpoints_support_backup.json +++ b/litellm/provider_endpoints_support_backup.json @@ -177,6 +177,23 @@ "interactions": true } }, + "apodex": { + "display_name": "Apodex (`apodex`)", + "url": "https://docs.litellm.ai/docs/providers/apodex", + "endpoints": { + "chat_completions": true, + "messages": true, + "responses": true, + "embeddings": false, + "image_generations": false, + "audio_transcriptions": false, + "audio_speech": false, + "moderations": false, + "batches": false, + "rerank": false, + "a2a": false + } + }, "apertis": { "display_name": "Apertis (`apertis`)", "endpoints": { diff --git a/tests/test_litellm/llms/apodex/test_apodex_chat_transformation.py b/tests/test_litellm/llms/apodex/test_apodex_chat_transformation.py index da31464d5cb..2655a14aae7 100644 --- a/tests/test_litellm/llms/apodex/test_apodex_chat_transformation.py +++ b/tests/test_litellm/llms/apodex/test_apodex_chat_transformation.py @@ -156,6 +156,17 @@ class TestStreamDefault: assert captured["body"]["stream"] is False assert captured["body"]["mcp_servers"] == mcp_servers + def test_extra_body_cannot_override_non_streaming_pin(self): + captured: dict = {} + litellm.completion( + model=DEEP_RESEARCH_MODEL, + messages=[{"role": "user", "content": "hi"}], + extra_body={"stream": True}, + client=_client(captured), + ) + + assert captured["body"]["stream"] is False + class TestSupportedParams: def test_core_models_support_tools(self): diff --git a/tests/test_litellm/llms/apodex/test_apodex_common_utils.py b/tests/test_litellm/llms/apodex/test_apodex_common_utils.py index 2da109c6ec4..da908d4d8e8 100644 --- a/tests/test_litellm/llms/apodex/test_apodex_common_utils.py +++ b/tests/test_litellm/llms/apodex/test_apodex_common_utils.py @@ -82,6 +82,12 @@ class TestRegistration: assert litellm.ApodexChatConfig().custom_llm_provider == "apodex" assert litellm.ApodexResponsesConfig().custom_llm_provider == LlmProviders.APODEX + def test_packaged_endpoint_matrix_matches_the_source(self): + source = json.loads((REPO_ROOT / "provider_endpoints_support.json").read_text()) + backup = json.loads((REPO_ROOT / "litellm" / "provider_endpoints_support_backup.json").read_text()) + + assert backup["providers"]["apodex"] == source["providers"]["apodex"] + class TestModelMetadata: @pytest.fixture(scope="class") diff --git a/tests/test_litellm/llms/apodex/test_apodex_messages_transformation.py b/tests/test_litellm/llms/apodex/test_apodex_messages_transformation.py index 09427b31a2f..0e4fb368cc5 100644 --- a/tests/test_litellm/llms/apodex/test_apodex_messages_transformation.py +++ b/tests/test_litellm/llms/apodex/test_apodex_messages_transformation.py @@ -62,12 +62,37 @@ class TestNativePassthroughRouting: forwarding to a path Apodex does not serve for these tiers.""" assert _messages_config(model) is None - def test_deep_research_translation_uses_responses_api(self): - from litellm.llms.anthropic.experimental_pass_through.messages.handler import ( - _should_route_to_responses_api, - ) + @pytest.mark.parametrize("stream", (False, True), ids=("non-streaming", "streaming")) + def test_deep_research_translation_uses_responses_api(self, monkeypatch: pytest.MonkeyPatch, stream: bool): + from litellm.llms.anthropic.experimental_pass_through.messages import handler - assert _should_route_to_responses_api("apodex") is True + captured: dict = {} + + class ResponsesRouteSelected(Exception): + pass + + def capture_responses_translation(**kwargs): + captured.update(kwargs) + raise ResponsesRouteSelected + + def reject_chat_translation(**kwargs): + pytest.fail("Apodex Deep Research messages must not route through chat completions") + + monkeypatch.setattr(litellm, "responses", capture_responses_translation) + monkeypatch.setattr(litellm, "completion", reject_chat_translation) + + with pytest.raises(ResponsesRouteSelected): + handler.anthropic_messages_handler( + max_tokens=256, + messages=[{"role": "user", "content": "hi"}], + model="apodex/apodex-1-1-deep-research", + custom_llm_provider="apodex", + stream=stream, + ) + + assert captured["model"] == "apodex-1-1-deep-research" + assert captured["custom_llm_provider"] == "apodex" + assert captured.get("stream", False) is stream class TestNativePassthroughRequest: From 42dcfa12a23688612b4c3d604d6b75ded9545010 Mon Sep 17 00:00:00 2001 From: zhanghanduo Date: Sun, 16 Aug 2026 19:47:24 +0800 Subject: [PATCH 08/12] fix(apodex): isolate chat routing --- litellm/llms/apodex/chat/transformation.py | 25 +++++++++---------- litellm/llms/apodex/common_utils.py | 5 ++++ .../apodex/test_apodex_chat_transformation.py | 25 +++++++++++++++++++ 3 files changed, 42 insertions(+), 13 deletions(-) diff --git a/litellm/llms/apodex/chat/transformation.py b/litellm/llms/apodex/chat/transformation.py index a1b13633f66..5b7e1e8ee6b 100644 --- a/litellm/llms/apodex/chat/transformation.py +++ b/litellm/llms/apodex/chat/transformation.py @@ -14,6 +14,7 @@ Ref: https://platform.apodex.ai/docs/chat-completions from collections.abc import Mapping from typing import Final +import litellm from litellm.llms.openai.chat.gpt_transformation import OpenAIGPTConfig from litellm.types.llms.openai import AllMessageValues @@ -22,6 +23,7 @@ from ..common_utils import ( get_apodex_api_base, get_apodex_api_key, is_deep_research_model, + is_responses_only_model, ) _DEEP_RESEARCH_PARAMS: Final = ( @@ -87,6 +89,13 @@ class ApodexChatConfig(OpenAIGPTConfig): model: str, drop_params: bool, ) -> dict: # mutable-ok: matches the base-class signature + if is_responses_only_model(model): + raise litellm.BadRequestError( + message=f"apodex model {model} is only available through /v1/responses", + model=model, + llm_provider="apodex", + ) + mapped: Final = super().map_openai_params( non_default_params=non_default_params, optional_params=optional_params, @@ -94,7 +103,6 @@ class ApodexChatConfig(OpenAIGPTConfig): drop_params=drop_params, ) - # Apodex documents max_tokens only. renamed: Final = ( mapped if "max_completion_tokens" not in mapped @@ -109,18 +117,9 @@ class ApodexChatConfig(OpenAIGPTConfig): if renamed.get("stream"): return renamed - # The OpenAI chat handler pops `stream` out of the params it forwards - # (litellm/llms/openai/openai.py), which would leave the Deep Research tiers - # on their streaming default. extra_body is merged into the request body - # further down, so it survives that pop. - requested_extra_body: Final = renamed.get("extra_body") - extra_body: Final = ( - requested_extra_body if isinstance(requested_extra_body, Mapping) else {} # mutable-ok: JSON request body - ) return { # mutable-ok: JSON request body **renamed, _PIN_NON_STREAMING: True, - "extra_body": {**extra_body, "stream": False}, # mutable-ok: JSON request body } def transform_request( @@ -131,12 +130,12 @@ class ApodexChatConfig(OpenAIGPTConfig): litellm_params: dict, # mutable-ok: matches the base-class signature headers: dict, # mutable-ok: matches the base-class signature ) -> dict: # mutable-ok: JSON request body - """Apply the non-streaming pin after LiteLLM merges caller ``extra_body``.""" - pin_non_streaming: Final = bool(optional_params.pop(_PIN_NON_STREAMING, False)) + pin_non_streaming: Final = bool(optional_params.get(_PIN_NON_STREAMING, False)) + forwarded_params: Final = {key: value for key, value in optional_params.items() if key != _PIN_NON_STREAMING} transformed: Final = super().transform_request( model=model, messages=messages, - optional_params=optional_params, + optional_params=forwarded_params, litellm_params=litellm_params, headers=headers, ) diff --git a/litellm/llms/apodex/common_utils.py b/litellm/llms/apodex/common_utils.py index ad29673ff4b..12dcb1414b6 100644 --- a/litellm/llms/apodex/common_utils.py +++ b/litellm/llms/apodex/common_utils.py @@ -17,6 +17,7 @@ from litellm.secret_managers.main import get_secret_str APODEX_API_BASE_URL: Final = "https://api.apodex.ai/v1" _DEEP_RESEARCH_MARKER: Final = "-deep-" +_RESPONSES_ONLY_MODELS: Final = frozenset({"apodex-1-1-deep-discover"}) def strip_provider_prefix(model: str) -> str: @@ -28,6 +29,10 @@ def is_deep_research_model(model: str) -> bool: return _DEEP_RESEARCH_MARKER in strip_provider_prefix(model) +def is_responses_only_model(model: str) -> bool: + return strip_provider_prefix(model) in _RESPONSES_ONLY_MODELS + + def get_apodex_api_key(api_key: str | None = None) -> str | None: return api_key or get_secret_str("APODEX_API_KEY") diff --git a/tests/test_litellm/llms/apodex/test_apodex_chat_transformation.py b/tests/test_litellm/llms/apodex/test_apodex_chat_transformation.py index 2655a14aae7..7cfbefca047 100644 --- a/tests/test_litellm/llms/apodex/test_apodex_chat_transformation.py +++ b/tests/test_litellm/llms/apodex/test_apodex_chat_transformation.py @@ -167,8 +167,33 @@ class TestStreamDefault: assert captured["body"]["stream"] is False + def test_transform_does_not_mutate_optional_params(self): + config = _chat_config("apodex-1.1") + optional_params = config.map_openai_params( + non_default_params={}, optional_params={}, model="apodex-1.1", drop_params=False + ) + original = optional_params.copy() + + config.transform_request( + model="apodex-1.1", + messages=[{"role": "user", "content": "hi"}], + optional_params=optional_params, + litellm_params={"custom_llm_provider": "apodex"}, + headers={}, + ) + + assert optional_params == original + class TestSupportedParams: + def test_responses_only_model_rejects_chat_completions(self): + with pytest.raises(litellm.BadRequestError, match="only available through /v1/responses"): + litellm.completion( + model="apodex/apodex-1-1-deep-discover", + messages=[{"role": "user", "content": "hi"}], + client=_client({}), + ) + def test_core_models_support_tools(self): supported = _chat_config("apodex-1.1").get_supported_openai_params("apodex-1.1") assert "tools" in supported From ec3e293ac2a073ec8de7080e6951fd7f326ea623 Mon Sep 17 00:00:00 2001 From: zhanghanduo Date: Sun, 16 Aug 2026 20:26:47 +0800 Subject: [PATCH 09/12] feat(apodex): stream Deep Research reasoning as reasoning deltas A Deep Research run streams two agents. The worker emits its chain of thought on the `reasoning` channel and a draft answer on a channel-less delta; the reporter emits its own reasoning plus the single `output_text` delta that matches the final response.completed snapshot. Only `output_text` was mapped, so 176 of 181 deltas in a sample run surfaced as GenericEvent and the reasoning was effectively lost. Map the `reasoning` channel to response.reasoning_summary_text.delta, which LiteLLM already translates into an Anthropic thinking_delta on the /v1/messages route Deep Research takes, and give it its own item id. The channel-less deltas stay unclaimed on purpose: splicing the worker's draft into the answer would corrupt the text. The remaining response.swarm.* lifecycle events keep passing through, since transform_streaming_response has no way to drop a chunk and run_finished carries the final content. --- .../llms/apodex/responses/transformation.py | 89 ++++++++++++------- .../test_apodex_responses_transformation.py | 70 +++++++++++++-- 2 files changed, 123 insertions(+), 36 deletions(-) diff --git a/litellm/llms/apodex/responses/transformation.py b/litellm/llms/apodex/responses/transformation.py index 065b7a475ec..c91ca316752 100644 --- a/litellm/llms/apodex/responses/transformation.py +++ b/litellm/llms/apodex/responses/transformation.py @@ -19,7 +19,8 @@ from __future__ import annotations from collections.abc import Mapping from time import time -from typing import TYPE_CHECKING, Final +from types import MappingProxyType +from typing import TYPE_CHECKING, Final, NamedTuple import httpx from pydantic import TypeAdapter, ValidationError @@ -45,6 +46,27 @@ _STATEFUL_PARAMS: Final = ("previous_response_id", "background") _CANCEL_RESPONSE_ADAPTER: Final = TypeAdapter(dict[str, object]) _BODY_FRAMING_HEADERS: Final = frozenset({"content-encoding", "content-length"}) +_SWARM_DELTA_EVENT: Final = "response.swarm.llm_delta" + + +class _SwarmChannel(NamedTuple): + """How one `swarm.data.channel` maps onto the OpenAI event that carries it.""" + + event_type: str + item_id_prefix: str + index_field: str + + +# A Deep Research run streams several agents. Only the reporter's `output_text` is +# the answer that lands in the final `response.completed` snapshot; the worker's +# channel-less deltas are an intermediate draft and must not be mistaken for it. +_SWARM_CHANNELS: Final = MappingProxyType( + { + "output_text": _SwarmChannel("response.output_text.delta", "msg", "content_index"), + "reasoning": _SwarmChannel("response.reasoning_summary_text.delta", "rs", "summary_index"), + } +) + class ApodexResponsesConfig(OpenAIResponsesAPIConfig): @property @@ -115,42 +137,49 @@ class ApodexResponsesConfig(OpenAIResponsesAPIConfig): parsed_chunk: dict, # mutable-ok: matches the base-class signature logging_obj: LiteLLMLoggingObj, ) -> ResponsesAPIStreamingResponse: - """Surface the Deep Research answer text as the OpenAI delta event callers expect. + """Surface a Deep Research run's text as the OpenAI delta events callers expect. - Observed live, not documented: a Deep Research stream carries its text in - `response.swarm.llm_delta` and never emits `response.output_text.delta`, so - without this the answer arrives only in the final `response.completed` - snapshot. `channel` splits the agent's reasoning from its answer; everything - else falls through to the base class as a GenericEvent. + Observed live, not documented: the stream carries all of its text in + `response.swarm.llm_delta` and never emits `response.output_text.delta` or + any reasoning event, so without this the answer arrives only in the final + `response.completed` snapshot and the reasoning is lost. Everything this + does not recognise, the remaining `response.swarm.*` lifecycle events + included, falls through to the base class as a GenericEvent. """ - swarm: Final = parsed_chunk.get("swarm") - swarm_data: Final = swarm.get("data") if isinstance(swarm, dict) else None - if ( - parsed_chunk.get("type") != "response.swarm.llm_delta" - or not isinstance(swarm_data, dict) - or swarm_data.get("channel") != "output_text" - or not isinstance(swarm_data.get("delta"), str) - ): - return super().transform_streaming_response( - model=model, - parsed_chunk=parsed_chunk, - logging_obj=logging_obj, - ) - - response_id: Final = str(parsed_chunk.get("response_id", "")) + mapped: Final = self._map_swarm_delta(parsed_chunk) return super().transform_streaming_response( model=model, - parsed_chunk={ # mutable-ok: JSON event payload - "type": "response.output_text.delta", - "item_id": f"msg_{response_id}", - "output_index": 0, - "content_index": 0, - "delta": swarm_data["delta"], - "sequence_number": parsed_chunk.get("sequence_number", 0), - }, + parsed_chunk=parsed_chunk if mapped is None else mapped, logging_obj=logging_obj, ) + @staticmethod + def _map_swarm_delta( + parsed_chunk: Mapping[str, object], + ) -> dict[str, object] | None: # mutable-ok: feeds the base class's `parsed_chunk: dict` + """The OpenAI event for this swarm delta, or None to pass the chunk through.""" + if parsed_chunk.get("type") != _SWARM_DELTA_EVENT: + return None + swarm: Final = parsed_chunk.get("swarm") + data: Final = swarm.get("data") if isinstance(swarm, Mapping) else None + if not isinstance(data, Mapping): + return None + + channel: Final = _SWARM_CHANNELS.get(data.get("channel")) + delta: Final = data.get("delta") + if channel is None or not isinstance(delta, str): + return None + + response_id: Final = str(parsed_chunk.get("response_id", "")) + return { # mutable-ok: JSON event payload + "type": channel.event_type, + "item_id": f"{channel.item_id_prefix}_{response_id}", + "output_index": 0, + channel.index_field: 0, + "delta": delta, + "sequence_number": parsed_chunk.get("sequence_number", 0), + } + def get_supported_openai_params(self, model: str) -> list: # mutable-ok: matches the base-class signature inherited: Final = super().get_supported_openai_params(model) if is_deep_research_model(model): diff --git a/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py b/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py index 789911126c3..e6bb831d2e5 100644 --- a/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py +++ b/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py @@ -246,7 +246,7 @@ class TestDeepResearchKeepsState: "response_id": "w_c4b77c96", "sequence_number": 12, "swarm": { - "agent_id": "stateful_react", + "agent_id": "reporter", "data": { "channel": "output_text", "delta": "Hello there, friend!", @@ -263,15 +263,46 @@ class TestDeepResearchKeepsState: assert event.item_id == "msg_w_c4b77c96" assert event.delta == "Hello there, friend!" assert event.sequence_number == 12 + assert event.content_index == 0 + + def test_reasoning_delta_becomes_a_reasoning_summary_delta(self): + """`response.reasoning_summary_text.delta` is what LiteLLM already translates + into an Anthropic `thinking_delta`, which is the route Deep Research takes on + /v1/messages. It also keeps a separate item id from the answer text.""" + config = _responses_config("apodex-1-1-deep-research") + event = config.transform_streaming_response( + model="apodex-1-1-deep-research", + parsed_chunk={ + "type": "response.swarm.llm_delta", + "response_id": "w_c4b77c96", + "sequence_number": 7, + "swarm": { + "agent_id": "stateful_react", + "data": {"channel": "reasoning", "delta": "The user wants", "delta_index": 0}, + }, + }, + logging_obj=None, + ) + + assert event.type == "response.reasoning_summary_text.delta" + assert event.item_id == "rs_w_c4b77c96" + assert event.delta == "The user wants" + assert event.summary_index == 0 + assert not hasattr(event, "content_index") @pytest.mark.parametrize( "channel", - ("reasoning", None), - ids=("reasoning-channel", "no-channel"), + (None, "tool_output"), + ids=("no-channel", "unknown-channel"), ) - def test_non_answer_deltas_are_not_claimed_as_output_text(self, channel): - """Most of the stream is the agent thinking; only `output_text` is the answer.""" - data = {"delta": "The"} if channel is None else {"channel": channel, "delta": "The"} + def test_intermediate_agent_deltas_are_not_claimed(self, channel): + """The worker agent streams a draft answer on a channel-less delta. + + Live capture: those four deltas spell "Hello, friend! How are you?" while the + reporter's `output_text` is the "Hello there, friend!" that lands in + response.completed. Claiming them would splice the draft into the answer. + """ + data = {"delta": "Hello,"} if channel is None else {"channel": channel, "delta": "Hello,"} config = _responses_config("apodex-1-1-deep-research") event = config.transform_streaming_response( model="apodex-1-1-deep-research", @@ -286,6 +317,33 @@ class TestDeepResearchKeepsState: assert event.type == "response.swarm.llm_delta" + @pytest.mark.parametrize( + "event_type", + ( + "response.swarm.run_started", + "response.swarm.run_finished", + "response.swarm.injection_window", + "response.swarm.llm_attempt_started", + "response.swarm.llm_attempt_finished", + ), + ) + def test_swarm_lifecycle_events_pass_through(self, event_type: str): + """LiteLLM cannot drop a chunk from the stream, so these stay as GenericEvent + rather than being silently swallowed; `run_finished` carries the final content.""" + config = _responses_config("apodex-1-1-deep-research") + event = config.transform_streaming_response( + model="apodex-1-1-deep-research", + parsed_chunk={ + "type": event_type, + "response_id": "w_c4b77c96", + "sequence_number": 4, + "swarm": {"agent_id": "reporter", "data": {"status": "success"}}, + }, + logging_obj=None, + ) + + assert event.type == event_type + def test_documented_events_pass_through_untouched(self): config = _responses_config("apodex-1-1-deep-research") event = config.transform_streaming_response( From 428380fb1e18050ecaf64536fbfa81f35f8b663b Mon Sep 17 00:00:00 2001 From: zhanghanduo Date: Sun, 16 Aug 2026 20:47:02 +0800 Subject: [PATCH 10/12] fix(apodex): preserve reasoning block lifecycle --- .../responses_adapters/streaming_iterator.py | 121 ++++++++++-------- .../llms/apodex/responses/transformation.py | 5 +- ...t_responses_adapters_streaming_iterator.py | 55 +++++++- .../test_apodex_responses_transformation.py | 54 +++++++- 4 files changed, 177 insertions(+), 58 deletions(-) diff --git a/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py b/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py index f12dd979338..8b9a3ed6764 100644 --- a/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py +++ b/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py @@ -3,7 +3,7 @@ import json import traceback from collections import deque -from collections.abc import AsyncIterator +from collections.abc import AsyncIterator, Mapping from typing import Any, Final from litellm import verbose_logger @@ -36,6 +36,8 @@ class AnthropicResponsesStreamWrapper: self.model = model self._message_id: str = f"msg_{uuid.uuid4()}" self._current_block_index: int = -1 + self._open_block_index: int | None = None + self._open_block_type: str | None = None # Map item_id -> content_block_index so we can stop the right block later self._item_id_to_block_index: dict[str, int] = {} # Track open function_call items by item_id so we can emit tool_use start @@ -68,6 +70,48 @@ class AnthropicResponsesStreamWrapper: self._current_block_index += 1 return self._current_block_index + def _close_open_block(self) -> None: + if self._open_block_index is None: + return + self._chunk_queue.append( + { + "type": "content_block_stop", + "index": self._open_block_index, + } + ) + self._open_block_index = None + self._open_block_type = None + + def _start_block(self, block_idx: int, block_type: str, content_block: Mapping[str, object]) -> None: + self._close_open_block() + self._chunk_queue.append( + { + "type": "content_block_start", + "index": block_idx, + "content_block": dict(content_block), + } + ) + self._open_block_index = block_idx + self._open_block_type = block_type + + def _get_or_start_block( + self, + item_id: str | None, + block_type: str, + content_block: Mapping[str, object], + ) -> int: + mapped_index: Final = self._item_id_to_block_index.get(item_id) if item_id else None + if mapped_index is not None: + return mapped_index + if item_id is None and self._open_block_index is not None and self._open_block_type == block_type: + return self._open_block_index + + block_idx: Final = self._next_block_index() + if item_id: + self._item_id_to_block_index[item_id] = block_idx + self._start_block(block_idx, block_type, content_block) + return block_idx + def _process_event(self, event: Any) -> None: """Convert one Responses API event into zero or more Anthropic chunks queued for emission.""" event_type = getattr(event, "type", None) @@ -96,13 +140,7 @@ class AnthropicResponsesStreamWrapper: block_idx = self._next_block_index() if item_id: self._item_id_to_block_index[item_id] = block_idx - self._chunk_queue.append( - { - "type": "content_block_start", - "index": block_idx, - "content_block": {"type": "text", "text": ""}, - } - ) + self._start_block(block_idx, "text", {"type": "text", "text": ""}) elif item_type == "function_call": call_id: Final = ( getattr(item, "call_id", None) or (item.get("call_id") if isinstance(item, dict) else None) or "" @@ -112,54 +150,36 @@ class AnthropicResponsesStreamWrapper: if item_id: self._item_id_to_block_index[item_id] = block_idx self._pending_tool_ids[item_id] = call_id - self._chunk_queue.append( + self._start_block( + block_idx, + "tool_use", { - "type": "content_block_start", - "index": block_idx, - "content_block": { - "type": "tool_use", - "id": call_id, - "name": name, - "input": {}, - }, - } + "type": "tool_use", + "id": call_id, + "name": name, + "input": {}, + }, ) elif item_type == "reasoning": block_idx = self._next_block_index() if item_id: self._item_id_to_block_index[item_id] = block_idx - self._chunk_queue.append( - { - "type": "content_block_start", - "index": block_idx, - "content_block": {"type": "thinking", "thinking": ""}, - } - ) + self._start_block(block_idx, "thinking", {"type": "thinking", "thinking": ""}) return # ---- text delta ---- if event_type == "response.output_text.delta": item_id = getattr(event, "item_id", None) or (event.get("item_id") if isinstance(event, dict) else None) delta = getattr(event, "delta", "") or (event.get("delta", "") if isinstance(event, dict) else "") - block_idx = self._item_id_to_block_index.get(item_id, -1) if item_id else self._current_block_index - if block_idx < 0: - # Some providers (e.g. LMStudio) skip response.output_item.added, - # so no text block is open yet; synthesize content_block_start - # instead of emitting a delta with index -1 - block_idx = self._next_block_index() - if item_id: - self._item_id_to_block_index[item_id] = block_idx - self._chunk_queue.append( - { - "type": "content_block_start", - "index": block_idx, - "content_block": {"type": "text", "text": ""}, - } - ) + text_block_idx: Final = self._get_or_start_block( + item_id=item_id, + block_type="text", + content_block={"type": "text", "text": ""}, + ) self._chunk_queue.append( { "type": "content_block_delta", - "index": block_idx, + "index": text_block_idx, "delta": {"type": "text_delta", "text": delta}, } ) @@ -169,15 +189,15 @@ class AnthropicResponsesStreamWrapper: if event_type == "response.reasoning_summary_text.delta": item_id = getattr(event, "item_id", None) or (event.get("item_id") if isinstance(event, dict) else None) delta = getattr(event, "delta", "") or (event.get("delta", "") if isinstance(event, dict) else "") - block_idx = ( - self._item_id_to_block_index.get(item_id, self._current_block_index) - if item_id - else self._current_block_index + thinking_block_idx: Final = self._get_or_start_block( + item_id=item_id, + block_type="thinking", + content_block={"type": "thinking", "thinking": ""}, ) self._chunk_queue.append( { "type": "content_block_delta", - "index": block_idx, + "index": thinking_block_idx, "delta": {"type": "thinking_delta", "thinking": delta}, } ) @@ -212,12 +232,8 @@ class AnthropicResponsesStreamWrapper: if item_id else self._current_block_index ) - self._chunk_queue.append( - { - "type": "content_block_stop", - "index": block_idx, - } - ) + if block_idx == self._open_block_index: + self._close_open_block() return # ---- response completed -> message_delta + message_stop ---- @@ -226,6 +242,7 @@ class AnthropicResponsesStreamWrapper: "response.failed", "response.incomplete", ): + self._close_open_block() response_obj: Final = getattr(event, "response", None) or ( event.get("response") if isinstance(event, dict) else None ) diff --git a/litellm/llms/apodex/responses/transformation.py b/litellm/llms/apodex/responses/transformation.py index c91ca316752..1e6d2fbd1f4 100644 --- a/litellm/llms/apodex/responses/transformation.py +++ b/litellm/llms/apodex/responses/transformation.py @@ -165,8 +165,11 @@ class ApodexResponsesConfig(OpenAIResponsesAPIConfig): if not isinstance(data, Mapping): return None - channel: Final = _SWARM_CHANNELS.get(data.get("channel")) + channel_name: Final = data.get("channel") delta: Final = data.get("delta") + if not isinstance(channel_name, str): + return None + channel: Final = _SWARM_CHANNELS.get(channel_name) if channel is None or not isinstance(delta, str): return None diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py index 73b58e71009..f8c34c9bf7d 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py @@ -113,9 +113,10 @@ class TestProcessEventTextDeltaWithoutOutputItemAdded: {"type": "response.output_text.delta", "item_id": "m1", "delta": "Hi"}, ] ) - assert chunks[1]["type"] == "content_block_start" - assert chunks[1]["content_block"] == {"type": "text", "text": ""} - assert [c["index"] for c in chunks[1:]] == [1, 1] + assert chunks[1] == {"type": "content_block_stop", "index": 0} + assert chunks[2]["type"] == "content_block_start" + assert chunks[2]["content_block"] == {"type": "text", "text": ""} + assert [c["index"] for c in chunks[1:]] == [0, 1, 1] def test_process_event_registered_item_id_does_not_synthesize_start(self): chunks = _process_all( @@ -133,6 +134,54 @@ class TestProcessEventTextDeltaWithoutOutputItemAdded: ] +class TestProcessEventReasoningDeltaWithoutOutputItemAdded: + def test_reasoning_opens_thinking_block_before_delta(self): + chunks = _process_all( + [ + {"type": "response.reasoning_summary_text.delta", "item_id": "rs_1", "delta": "Think "}, + {"type": "response.reasoning_summary_text.delta", "item_id": "rs_1", "delta": "carefully"}, + ] + ) + + assert [chunk["type"] for chunk in chunks] == [ + "content_block_start", + "content_block_delta", + "content_block_delta", + ] + assert chunks[0] == { + "type": "content_block_start", + "index": 0, + "content_block": {"type": "thinking", "thinking": ""}, + } + assert chunks[1] == { + "type": "content_block_delta", + "index": 0, + "delta": {"type": "thinking_delta", "thinking": "Think "}, + } + + def test_reasoning_and_text_get_separate_closed_blocks(self): + response = SimpleNamespace(status="completed", output=[], usage=None) + chunks = _process_all( + [ + {"type": "response.reasoning_summary_text.delta", "item_id": "rs_1", "delta": "Think"}, + {"type": "response.output_text.delta", "item_id": "msg_1", "delta": "Answer"}, + {"type": "response.completed", "response": response}, + ] + ) + + assert [(chunk["type"], chunk.get("index")) for chunk in chunks] == [ + ("content_block_start", 0), + ("content_block_delta", 0), + ("content_block_stop", 0), + ("content_block_start", 1), + ("content_block_delta", 1), + ("content_block_stop", 1), + ("message_delta", None), + ("message_stop", None), + ] + assert chunks[3]["content_block"] == {"type": "text", "text": ""} + + class TestResponseCompletedUsage: """The Anthropic ``message_delta`` usage must report cache reads/writes and exclude them from ``input_tokens``, so spend is not billed at the uncached diff --git a/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py b/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py index e6bb831d2e5..032fb99db67 100644 --- a/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py +++ b/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py @@ -7,11 +7,15 @@ the model rather than applied provider-wide. """ import gzip +from types import SimpleNamespace import httpx import pytest import litellm +from litellm.llms.anthropic.experimental_pass_through.responses_adapters.streaming_iterator import ( + AnthropicResponsesStreamWrapper, +) from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler from litellm.types.utils import LlmProviders from litellm.utils import ProviderConfigManager @@ -290,10 +294,56 @@ class TestDeepResearchKeepsState: assert event.summary_index == 0 assert not hasattr(event, "content_index") + def test_reasoning_and_answer_form_valid_anthropic_blocks(self): + config = _responses_config("apodex-1-1-deep-research") + reasoning_event = config.transform_streaming_response( + model="apodex-1-1-deep-research", + parsed_chunk={ + "type": "response.swarm.llm_delta", + "response_id": "w_c4b77c96", + "sequence_number": 7, + "swarm": {"data": {"channel": "reasoning", "delta": "The user wants"}}, + }, + logging_obj=None, + ) + answer_event = config.transform_streaming_response( + model="apodex-1-1-deep-research", + parsed_chunk={ + "type": "response.swarm.llm_delta", + "response_id": "w_c4b77c96", + "sequence_number": 8, + "swarm": {"data": {"channel": "output_text", "delta": "Hello there, friend!"}}, + }, + logging_obj=None, + ) + wrapper = AnthropicResponsesStreamWrapper(responses_stream=None, model="apodex-1-1-deep-research") + for event in ( + reasoning_event, + answer_event, + {"type": "response.completed", "response": SimpleNamespace(status="completed", output=[], usage=None)}, + ): + wrapper._process_event(event) + + chunks = list(wrapper._chunk_queue) + assert [(chunk["type"], chunk.get("index")) for chunk in chunks] == [ + ("content_block_start", 0), + ("content_block_delta", 0), + ("content_block_stop", 0), + ("content_block_start", 1), + ("content_block_delta", 1), + ("content_block_stop", 1), + ("message_delta", None), + ("message_stop", None), + ] + assert chunks[0]["content_block"]["type"] == "thinking" + assert chunks[1]["delta"] == {"type": "thinking_delta", "thinking": "The user wants"} + assert chunks[3]["content_block"]["type"] == "text" + assert chunks[4]["delta"] == {"type": "text_delta", "text": "Hello there, friend!"} + @pytest.mark.parametrize( "channel", - (None, "tool_output"), - ids=("no-channel", "unknown-channel"), + (None, "tool_output", []), + ids=("no-channel", "unknown-channel", "non-string-channel"), ) def test_intermediate_agent_deltas_are_not_claimed(self, channel): """The worker agent streams a draft answer on a channel-less delta. From a40b9983c55039485e65258fb6b75004a3c67816 Mon Sep 17 00:00:00 2001 From: zhanghanduo Date: Sun, 16 Aug 2026 21:25:36 +0800 Subject: [PATCH 11/12] fix(anthropic): reopen a content block when a resumed item id returns _get_or_start_block trusted the item_id -> block index map without checking whether that block was still open, so a provider that reuses one item id for a whole run and interleaves channels got a delta addressed to a stopped block. Replaying reasoning, text, reasoning, text produced content_block_delta index=0 after content_block_stop index=0, which is not a valid Anthropic stream. Treat the mapping as valid only while it points at the open block, and rebind the item to a fresh block otherwise. Items registered through response.output_item.added still reuse their block, since that block is the open one while its deltas arrive. Apodex Deep Research is what surfaced this: it labels every reasoning delta of a run rs_ and every answer delta msg_, so any interleaving hits the stale mapping. --- .../responses_adapters/streaming_iterator.py | 5 ++- ...t_responses_adapters_streaming_iterator.py | 37 +++++++++++++++++++ 2 files changed, 41 insertions(+), 1 deletion(-) diff --git a/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py b/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py index 8b9a3ed6764..3e6e6df4565 100644 --- a/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py +++ b/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py @@ -101,8 +101,11 @@ class AnthropicResponsesStreamWrapper: content_block: Mapping[str, object], ) -> int: mapped_index: Final = self._item_id_to_block_index.get(item_id) if item_id else None - if mapped_index is not None: + if mapped_index is not None and mapped_index == self._open_block_index: return mapped_index + # A resumed item whose block already closed needs a fresh one: Anthropic rejects + # a delta addressed to a stopped block. Providers that reuse one item id for a + # whole run, then interleave channels, land here. if item_id is None and self._open_block_index is not None and self._open_block_type == block_type: return self._open_block_index diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py index f8c34c9bf7d..81b207bb259 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py @@ -181,6 +181,43 @@ class TestProcessEventReasoningDeltaWithoutOutputItemAdded: ] assert chunks[3]["content_block"] == {"type": "text", "text": ""} + def test_resumed_item_id_opens_a_fresh_block(self): + """A provider that reuses one item id per run, then interleaves channels, would + otherwise address a delta to a block that has already stopped.""" + response = SimpleNamespace(status="completed", output=[], usage=None) + chunks = _process_all( + [ + {"type": "response.reasoning_summary_text.delta", "item_id": "rs_1", "delta": "Think A"}, + {"type": "response.output_text.delta", "item_id": "msg_1", "delta": "Answer A"}, + {"type": "response.reasoning_summary_text.delta", "item_id": "rs_1", "delta": "Think B"}, + {"type": "response.output_text.delta", "item_id": "msg_1", "delta": "Answer B"}, + {"type": "response.completed", "response": response}, + ] + ) + + assert [(chunk["type"], chunk.get("index")) for chunk in chunks] == [ + ("content_block_start", 0), + ("content_block_delta", 0), + ("content_block_stop", 0), + ("content_block_start", 1), + ("content_block_delta", 1), + ("content_block_stop", 1), + ("content_block_start", 2), + ("content_block_delta", 2), + ("content_block_stop", 2), + ("content_block_start", 3), + ("content_block_delta", 3), + ("content_block_stop", 3), + ("message_delta", None), + ("message_stop", None), + ] + assert [chunk["content_block"]["type"] for chunk in chunks if chunk["type"] == "content_block_start"] == [ + "thinking", + "text", + "thinking", + "text", + ] + class TestResponseCompletedUsage: """The Anthropic ``message_delta`` usage must report cache reads/writes and From a665d2e3367e59a3e9786abeab4b8e8500fb78c9 Mon Sep 17 00:00:00 2001 From: zhanghanduo Date: Wed, 26 Aug 2026 09:38:19 +0800 Subject: [PATCH 12/12] fix(apodex): satisfy CI quality gates --- litellm/__init__.py | 2 +- .../responses_adapters/streaming_iterator.py | 18 +++++++++++++----- litellm/llms/apodex/chat/transformation.py | 6 ++++-- .../llms/apodex/responses/transformation.py | 6 +++--- ...st_responses_adapters_streaming_iterator.py | 8 +++++++- .../apodex/test_apodex_chat_transformation.py | 7 +++++++ .../test_apodex_messages_transformation.py | 1 + .../test_apodex_responses_transformation.py | 12 ++++++++++++ 8 files changed, 48 insertions(+), 12 deletions(-) diff --git a/litellm/__init__.py b/litellm/__init__.py index d5a6af8173f..e36e94e10d1 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -644,7 +644,7 @@ snowflake_models: Set = set() gradient_ai_models: Set = set() llama_models: Set = set() nscale_models: Set = set() -apodex_models: Set = set() +apodex_models: Set = set() # mutable-ok: provider model registry is populated during initialization nebius_models: Set = set() nebius_embedding_models: Set = set() aiml_models: Set = set() diff --git a/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py b/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py index 2a310fe4c1a..834d7437c43 100644 --- a/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py +++ b/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py @@ -74,7 +74,7 @@ class AnthropicResponsesStreamWrapper: if self._open_block_index is None: return self._chunk_queue.append( - { + { # mutable-ok: queued Anthropic event payload "type": "content_block_stop", "index": self._open_block_index, } @@ -88,7 +88,7 @@ class AnthropicResponsesStreamWrapper: { "type": "content_block_start", "index": block_idx, - "content_block": dict(content_block), + "content_block": dict(content_block), # mutable-ok: queued Anthropic event payload } ) self._open_block_index = block_idx @@ -143,7 +143,11 @@ class AnthropicResponsesStreamWrapper: block_idx = self._next_block_index() if item_id: self._item_id_to_block_index[item_id] = block_idx - self._start_block(block_idx, "text", {"type": "text", "text": ""}) + self._start_block( + block_idx, + "text", + {"type": "text", "text": ""}, # mutable-ok: Anthropic content block payload + ) elif item_type == "function_call": call_id: Final = ( getattr(item, "call_id", None) or (item.get("call_id") if isinstance(item, dict) else None) or "" @@ -172,7 +176,7 @@ class AnthropicResponsesStreamWrapper: text_block_idx: Final = self._get_or_start_block( item_id=item_id, block_type="text", - content_block={"type": "text", "text": ""}, + content_block={"type": "text", "text": ""}, # mutable-ok: Anthropic content block payload ) self._chunk_queue.append( { @@ -192,7 +196,11 @@ class AnthropicResponsesStreamWrapper: thinking_block_idx: Final = self._get_or_start_block( item_id=item_id, block_type="thinking", - content_block={"type": "thinking", "thinking": "", "signature": ""}, + content_block={ # mutable-ok: Anthropic content block payload + "type": "thinking", + "thinking": "", + "signature": "", + }, ) self._chunk_queue.append( { diff --git a/litellm/llms/apodex/chat/transformation.py b/litellm/llms/apodex/chat/transformation.py index 5b7e1e8ee6b..6287aedeebb 100644 --- a/litellm/llms/apodex/chat/transformation.py +++ b/litellm/llms/apodex/chat/transformation.py @@ -125,13 +125,15 @@ class ApodexChatConfig(OpenAIGPTConfig): def transform_request( self, model: str, - messages: list[AllMessageValues], + messages: list[AllMessageValues], # mutable-ok: matches the base-class signature optional_params: dict, # mutable-ok: matches the base-class signature litellm_params: dict, # mutable-ok: matches the base-class signature headers: dict, # mutable-ok: matches the base-class signature ) -> dict: # mutable-ok: JSON request body pin_non_streaming: Final = bool(optional_params.get(_PIN_NON_STREAMING, False)) - forwarded_params: Final = {key: value for key, value in optional_params.items() if key != _PIN_NON_STREAMING} + forwarded_params: Final = { # mutable-ok: base transformer requires a request dict + key: value for key, value in optional_params.items() if key != _PIN_NON_STREAMING + } transformed: Final = super().transform_request( model=model, messages=messages, diff --git a/litellm/llms/apodex/responses/transformation.py b/litellm/llms/apodex/responses/transformation.py index 1e6d2fbd1f4..86357e118be 100644 --- a/litellm/llms/apodex/responses/transformation.py +++ b/litellm/llms/apodex/responses/transformation.py @@ -117,13 +117,13 @@ class ApodexResponsesConfig(OpenAIResponsesAPIConfig): status_code=raw_response.status_code, # Content-Encoding and Content-Length describe the body being replaced here; # carrying them over makes httpx try to decompress plain JSON on read. - headers={ + headers={ # mutable-ok: httpx requires mutable response headers name: value for name, value in raw_response.headers.items() if name.lower() not in _BODY_FRAMING_HEADERS }, - json={ + json={ # mutable-ok: httpx requires a JSON-compatible response dict **payload, "created_at": payload.get("created_at", int(time())), - "output": payload.get("output", []), + "output": payload.get("output", []), # mutable-ok: Responses payload requires an array default }, ) return super().transform_cancel_response_api_response( diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py index d0da2056e84..28339143b72 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py @@ -245,10 +245,16 @@ class TestProcessEventTextDeltaWithoutOutputItemAdded: assert chunks[1]["delta"] == {"type": "text_delta", "text": "Hel"} def test_process_event_delta_without_item_id_never_yields_negative_index(self): - chunks = _process_all([{"type": "response.output_text.delta", "delta": "Hi"}]) + chunks = _process_all( + [ + {"type": "response.output_text.delta", "delta": "Hi"}, + {"type": "response.output_text.delta", "delta": " again"}, + ] + ) assert [(c["type"], c["index"]) for c in chunks] == [ ("content_block_start", 0), ("content_block_delta", 0), + ("content_block_delta", 0), ] def test_process_event_unregistered_item_id_opens_new_text_block(self): diff --git a/tests/test_litellm/llms/apodex/test_apodex_chat_transformation.py b/tests/test_litellm/llms/apodex/test_apodex_chat_transformation.py index 7cfbefca047..75bb3068ca3 100644 --- a/tests/test_litellm/llms/apodex/test_apodex_chat_transformation.py +++ b/tests/test_litellm/llms/apodex/test_apodex_chat_transformation.py @@ -72,6 +72,13 @@ def _chat_config(model: str): class TestProviderResolution: + def test_openai_compatible_provider_info_uses_apodex_credentials(self): + config = _chat_config("apodex-1.1") + assert config._get_openai_compatible_provider_info("https://override.test/v1", "sk-override") == ( + "https://override.test/v1", + "sk-override", + ) + def test_prefixed_model_resolves_to_the_default_base(self): model, provider, api_key, api_base = litellm.get_llm_provider(model=CORE_MODEL) assert (model, provider, api_key, api_base) == ( diff --git a/tests/test_litellm/llms/apodex/test_apodex_messages_transformation.py b/tests/test_litellm/llms/apodex/test_apodex_messages_transformation.py index 0e4fb368cc5..ed6da35bba6 100644 --- a/tests/test_litellm/llms/apodex/test_apodex_messages_transformation.py +++ b/tests/test_litellm/llms/apodex/test_apodex_messages_transformation.py @@ -55,6 +55,7 @@ class TestNativePassthroughRouting: assert config is not None assert type(config).__name__ == "ApodexAnthropicMessagesConfig" assert config.custom_llm_provider == "apodex" + assert config.should_strip_billing_metadata() is True @pytest.mark.parametrize("model", DEEP_RESEARCH_MODELS) def test_deep_research_models_fall_back_to_translation(self, model: str): diff --git a/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py b/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py index 032fb99db67..252a07bf607 100644 --- a/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py +++ b/tests/test_litellm/llms/apodex/test_apodex_responses_transformation.py @@ -367,6 +367,18 @@ class TestDeepResearchKeepsState: assert event.type == "response.swarm.llm_delta" + def test_non_string_delta_is_not_claimed(self): + config = _responses_config("apodex-1-1-deep-research") + assert ( + config._map_swarm_delta( + { + "type": "response.swarm.llm_delta", + "swarm": {"data": {"channel": "output_text", "delta": None}}, + } + ) + is None + ) + @pytest.mark.parametrize( "event_type", (