From d1526c09fccf5536fab360685e0868b0f81106b6 Mon Sep 17 00:00:00 2001 From: fenil modi Date: Sat, 3 Oct 2026 08:34:20 +0000 Subject: [PATCH 1/4] feat(aiand): add ai& as a JSON-configured provider Register ai& (https://api.aiand.com/v1) for chat completions, responses, and Anthropic-style messages via the openai_like JSON provider path, with AIAND_API_KEY/AIAND_API_BASE env vars and URL autodetection. Add all 13 catalog models to both cost maps with per-token input, output, and cache-read pricing, context and output limits, and capability flags sourced from https://api.aiand.com/v1/api.json, plus scripts/ sync_aiand_models.py and a daily workflow that opens a PR when the live catalog drifts from the registry. Removals stamp metadata. absent_from_spec_since so every finding is a real file change, matching the together_ai sync convention. Register the provider in the Add Model form, dashboard provider maps with logo, endpoint support matrices, and the README provider table. Cover provider resolution, credential override, cost invariants over all models, root/backup registry sync, respx-mocked chat, responses, and messages routing, and the sync script's add, update, removal, and reappearance paths in tests --- .github/workflows/sync-aiand-models.yml | 73 ++++ README.md | 1 + litellm/constants.py | 2 + litellm/llms/openai_like/providers.json | 6 + ...odel_prices_and_context_window_backup.json | 325 ++++++++++++++++++ .../provider_endpoints_support_backup.json | 18 + .../provider_create_fields.json | 28 ++ litellm/types/utils.py | 1 + model_prices_and_context_window.json | 325 ++++++++++++++++++ provider_endpoints_support.json | 18 + scripts/sync_aiand_models.py | 310 +++++++++++++++++ .../llms/openai_like/test_aiand_provider.py | 250 ++++++++++++++ tests/unit/test_sync_aiand_models.py | 256 ++++++++++++++ .../public/assets/logos/aiand.svg | 1 + .../components/provider_info_helpers.test.tsx | 17 + .../src/components/provider_info_helpers.tsx | 5 + 16 files changed, 1636 insertions(+) create mode 100644 .github/workflows/sync-aiand-models.yml create mode 100644 scripts/sync_aiand_models.py create mode 100644 tests/unit/llms/openai_like/test_aiand_provider.py create mode 100644 tests/unit/test_sync_aiand_models.py create mode 100644 ui/litellm-dashboard/public/assets/logos/aiand.svg diff --git a/.github/workflows/sync-aiand-models.yml b/.github/workflows/sync-aiand-models.yml new file mode 100644 index 00000000000..b476844b8d1 --- /dev/null +++ b/.github/workflows/sync-aiand-models.yml @@ -0,0 +1,73 @@ +name: Sync aiand model registry + +on: + schedule: + - cron: "45 7 * * *" + workflow_dispatch: + +concurrency: + group: sync-aiand-models + cancel-in-progress: false + +permissions: + contents: write + pull-requests: write + +jobs: + sync_aiand_models: + if: github.repository == 'BerriAI/litellm' + runs-on: ubuntu-latest + env: + BASE_BRANCH: ${{ github.event.repository.default_branch }} + steps: + - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + ref: ${{ env.BASE_BRANCH }} + persist-credentials: false + - name: Set up uv + uses: ./.github/actions/setup-uv-with-retries + with: + version: "0.10.9" + - name: Look for an already-open sync PR + id: existing + run: | + open_pr="$(gh pr list --repo "$GITHUB_REPOSITORY" --state open --limit 1000 --json headRefName,changedFiles \ + --jq '[.[] | select(.headRefName | startswith("litellm_aiand_registry_sync_")) | select(.changedFiles > 0)] | first | .headRefName // empty')" + echo "open_pr=$open_pr" >> "$GITHUB_OUTPUT" + if [ -n "$open_pr" ]; then + echo "Sync PR $open_pr still has unreviewed changes; skipping this run." + fi + env: + GH_TOKEN: ${{ secrets.GH_TOKEN || github.token }} + - name: Run the sync + if: steps.existing.outputs.open_pr == '' + run: | + uv run --frozen python scripts/sync_aiand_models.py --write --pr-body-file "$RUNNER_TEMP/pr_body.md" + - name: Regenerate the JSON schema + if: steps.existing.outputs.open_pr == '' + run: | + uv run --frozen python ci_cd/generate_model_prices_schema.py + - name: Create a pull request when the registry changed + if: steps.existing.outputs.open_pr == '' + run: | + if git diff --quiet; then + echo "Registry already in sync; no PR needed." + exit 0 + fi + branch="litellm_aiand_registry_sync_$(date +'%Y-%m-%d')" + git config user.name "github-actions[bot]" + git config user.email "41898282+github-actions[bot]@users.noreply.github.com" + git checkout -b "$branch" + git add model_prices_and_context_window.json \ + litellm/model_prices_and_context_window_backup.json \ + model_prices_and_context_window.schema.json + git commit -m "feat(models): sync aiand model registry $(date +'%Y-%m-%d')" + gh auth setup-git + git push origin --delete "$branch" 2>/dev/null || true + git push origin "$branch" + gh pr create --title "feat(models): sync aiand model registry" \ + --body-file "$RUNNER_TEMP/pr_body.md" \ + --head "$branch" \ + --base "$BASE_BRANCH" + env: + GH_TOKEN: ${{ secrets.GH_TOKEN || github.token }} diff --git a/README.md b/README.md index 4004e6474ee..a0b85c0a4b0 100644 --- a/README.md +++ b/README.md @@ -298,6 +298,7 @@ Set `LITELLM_PROXY_API_BASE` and `LITELLM_PROXY_API_KEY` and every model call th | Provider | `/chat/completions` | `/messages` | `/responses` | `/embeddings` | `/image/generations` | `/audio/transcriptions` | `/audio/speech` | `/moderations` | `/batches` | `/rerank` | |-------------------------------------------------------------------------------------|---------------------|-------------|--------------|---------------|----------------------|-------------------------|-----------------|----------------|-----------|-----------| | [Abliteration (`abliteration`)](https://docs.litellm.ai/docs/providers/abliteration) | ✅ | | | | | | | | | | +| [ai& (`aiand`)](https://docs.aiand.com/) | ✅ | ✅ | ✅ | | | | | | | | | [AI/ML API (`aiml`)](https://docs.litellm.ai/docs/providers/aiml) | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | | | [AI21 (`ai21`)](https://docs.litellm.ai/docs/providers/ai21) | ✅ | ✅ | ✅ | | | | | | | | | [AI21 Chat (`ai21_chat`)](https://docs.litellm.ai/docs/providers/ai21) | ✅ | ✅ | ✅ | | | | | | | | diff --git a/litellm/constants.py b/litellm/constants.py index 18fc6aa7e74..cd40ee22e57 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -939,6 +939,7 @@ openai_compatible_endpoints: Final[list] = [ "https://dashscope.aliyuncs.com/compatible-mode/v1", "https://api-inference.modelscope.cn/v1", "https://api.moonshot.ai/v1", + "https://api.aiand.com/v1", "https://api.publicai.co/v1", "https://api.synthetic.new/openai/v1", "https://serverless.tensormesh.ai/v1", @@ -1001,6 +1002,7 @@ openai_compatible_providers: Final[list] = [ "chatgpt", # ChatGPT subscription API "novita", "meta_llama", + "aiand", "publicai", # PublicAI - JSON-configured provider "synthetic", # Synthetic - JSON-configured provider "tensormesh", # Tensormesh - JSON-configured provider diff --git a/litellm/llms/openai_like/providers.json b/litellm/llms/openai_like/providers.json index 61ff4be3a46..f0d8f3828ec 100644 --- a/litellm/llms/openai_like/providers.json +++ b/litellm/llms/openai_like/providers.json @@ -107,6 +107,12 @@ "api_key_env": "AIHUBMIX_API_KEY", "api_base_env": "AIHUBMIX_API_BASE" }, + "aiand": { + "base_url": "https://api.aiand.com/v1", + "api_key_env": "AIAND_API_KEY", + "api_base_env": "AIAND_API_BASE", + "supported_endpoints": ["/v1/chat/completions", "/v1/responses", "/v1/messages"] + }, "crusoe": { "base_url": "https://managed-inference-api-proxy.crusoecloud.com/v1", "api_key_env": "CRUSOE_API_KEY", diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 7e8f9bb4093..dcd7477fa7d 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -59649,6 +59649,331 @@ ], "supports_audio_input": true }, + "aiand/deepseek-ai/deepseek-v4-flash": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 2.5e-07, + "cache_read_input_token_cost": 8e-08, + "max_input_tokens": 1048576, + "max_output_tokens": 384000, + "max_tokens": 384000, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/deepseek-ai/deepseek-v4-pro": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1e-06, + "output_cost_per_token": 2.5e-06, + "cache_read_input_token_cost": 2.5e-07, + "max_input_tokens": 1048576, + "max_output_tokens": 384000, + "max_tokens": 384000, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/deepseek-ai/deepseek-v4.1-flash": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 3e-07, + "output_cost_per_token": 6e-07, + "cache_read_input_token_cost": 2e-08, + "max_input_tokens": 1048576, + "max_output_tokens": 384000, + "max_tokens": 384000, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/google/gemma-4-31b-it": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 2e-07, + "output_cost_per_token": 5e-07, + "cache_read_input_token_cost": 5e-08, + "max_input_tokens": 262144, + "max_output_tokens": 32768, + "max_tokens": 32768, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/moonshotai/kimi-k2.7-code": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 7.5e-07, + "output_cost_per_token": 3.5e-06, + "cache_read_input_token_cost": 2e-07, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "max_tokens": 262144, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/moonshotai/kimi-k3": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 3e-06, + "output_cost_per_token": 1.25e-05, + "cache_read_input_token_cost": 5e-07, + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "max_tokens": 131072, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/motif-technologies/motif-3": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 5e-07, + "output_cost_per_token": 2e-06, + "cache_read_input_token_cost": 2e-07, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "max_tokens": 262144, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": false, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/openai/gpt-oss-120b": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 6e-07, + "cache_read_input_token_cost": 8e-08, + "max_input_tokens": 131072, + "max_output_tokens": 32768, + "max_tokens": 32768, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/qwen/qwen3.6-27b": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 3.2e-07, + "output_cost_per_token": 3.2e-06, + "cache_read_input_token_cost": 2e-07, + "max_input_tokens": 262144, + "max_output_tokens": 65536, + "max_tokens": 65536, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/qwen/qwen3.8-27b": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 4e-07, + "output_cost_per_token": 3e-06, + "cache_read_input_token_cost": 2e-07, + "max_input_tokens": 262144, + "max_output_tokens": 32768, + "max_tokens": 32768, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/zai-org/glm-5.2": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1e-06, + "output_cost_per_token": 4e-06, + "cache_read_input_token_cost": 3e-07, + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "max_tokens": 131072, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/zai-org/glm-5.3": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1e-06, + "output_cost_per_token": 4e-06, + "cache_read_input_token_cost": 3e-07, + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "max_tokens": 131072, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/zai-org/glm-5.3-flash": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 5e-07, + "cache_read_input_token_cost": 3e-08, + "max_input_tokens": 1048550, + "max_output_tokens": 131072, + "max_tokens": 131072, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, "tensormesh/Qwen/Qwen3.5-397B-A17B-FP8": { "litellm_provider": "tensormesh", "mode": "chat", diff --git a/litellm/provider_endpoints_support_backup.json b/litellm/provider_endpoints_support_backup.json index c9635587eeb..2694af5ada1 100644 --- a/litellm/provider_endpoints_support_backup.json +++ b/litellm/provider_endpoints_support_backup.json @@ -120,6 +120,24 @@ "interactions": true } }, + "aiand": { + "display_name": "ai& (`aiand`)", + "url": "https://docs.aiand.com/api/chat-completions/", + "endpoints": { + "chat_completions": true, + "messages": true, + "responses": true, + "embeddings": false, + "image_generations": false, + "audio_transcriptions": false, + "audio_speech": false, + "moderations": false, + "batches": false, + "rerank": false, + "a2a": false, + "interactions": false + } + }, "amazon_nova": { "display_name": "Amazon Nova (`amazon_nova`)", "url": "https://docs.litellm.ai/docs/providers/amazon_nova", diff --git a/litellm/proxy/public_endpoints/provider_create_fields.json b/litellm/proxy/public_endpoints/provider_create_fields.json index 6e96d6ad0ec..2a8995d0db1 100644 --- a/litellm/proxy/public_endpoints/provider_create_fields.json +++ b/litellm/proxy/public_endpoints/provider_create_fields.json @@ -1,4 +1,32 @@ [ + { + "provider": "AIAND", + "provider_display_name": "ai&", + "litellm_provider": "aiand", + "credential_fields": [ + { + "key": "api_base", + "label": "API Base", + "placeholder": "https://api.aiand.com/v1", + "tooltip": null, + "required": false, + "field_type": "text", + "options": null, + "default_value": null + }, + { + "key": "api_key", + "label": "API Key", + "placeholder": null, + "tooltip": null, + "required": true, + "field_type": "password", + "options": null, + "default_value": null + } + ], + "default_model_placeholder": "aiand/deepseek-ai/deepseek-v4.1-flash" + }, { "provider": "AIML", "provider_display_name": "AI/ML API", diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 6919fd6fd27..5a117b85c30 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -4153,6 +4153,7 @@ class LlmProviders(str, Enum): PARASAIL = "parasail" XIAOMI_MIMO = "xiaomi_mimo" TENSORMESH = "tensormesh" + AIAND = "aiand" LIBERTAI = "libertai" PINSTRIPES = "pinstripes" COGNITION = "cognition" diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 7e8f9bb4093..dcd7477fa7d 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -59649,6 +59649,331 @@ ], "supports_audio_input": true }, + "aiand/deepseek-ai/deepseek-v4-flash": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 2.5e-07, + "cache_read_input_token_cost": 8e-08, + "max_input_tokens": 1048576, + "max_output_tokens": 384000, + "max_tokens": 384000, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/deepseek-ai/deepseek-v4-pro": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1e-06, + "output_cost_per_token": 2.5e-06, + "cache_read_input_token_cost": 2.5e-07, + "max_input_tokens": 1048576, + "max_output_tokens": 384000, + "max_tokens": 384000, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/deepseek-ai/deepseek-v4.1-flash": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 3e-07, + "output_cost_per_token": 6e-07, + "cache_read_input_token_cost": 2e-08, + "max_input_tokens": 1048576, + "max_output_tokens": 384000, + "max_tokens": 384000, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/google/gemma-4-31b-it": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 2e-07, + "output_cost_per_token": 5e-07, + "cache_read_input_token_cost": 5e-08, + "max_input_tokens": 262144, + "max_output_tokens": 32768, + "max_tokens": 32768, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/moonshotai/kimi-k2.7-code": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 7.5e-07, + "output_cost_per_token": 3.5e-06, + "cache_read_input_token_cost": 2e-07, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "max_tokens": 262144, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/moonshotai/kimi-k3": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 3e-06, + "output_cost_per_token": 1.25e-05, + "cache_read_input_token_cost": 5e-07, + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "max_tokens": 131072, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/motif-technologies/motif-3": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 5e-07, + "output_cost_per_token": 2e-06, + "cache_read_input_token_cost": 2e-07, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "max_tokens": 262144, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": false, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/openai/gpt-oss-120b": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 6e-07, + "cache_read_input_token_cost": 8e-08, + "max_input_tokens": 131072, + "max_output_tokens": 32768, + "max_tokens": 32768, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/qwen/qwen3.6-27b": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 3.2e-07, + "output_cost_per_token": 3.2e-06, + "cache_read_input_token_cost": 2e-07, + "max_input_tokens": 262144, + "max_output_tokens": 65536, + "max_tokens": 65536, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/qwen/qwen3.8-27b": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 4e-07, + "output_cost_per_token": 3e-06, + "cache_read_input_token_cost": 2e-07, + "max_input_tokens": 262144, + "max_output_tokens": 32768, + "max_tokens": 32768, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/zai-org/glm-5.2": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1e-06, + "output_cost_per_token": 4e-06, + "cache_read_input_token_cost": 3e-07, + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "max_tokens": 131072, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/zai-org/glm-5.3": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1e-06, + "output_cost_per_token": 4e-06, + "cache_read_input_token_cost": 3e-07, + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "max_tokens": 131072, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/zai-org/glm-5.3-flash": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 5e-07, + "cache_read_input_token_cost": 3e-08, + "max_input_tokens": 1048550, + "max_output_tokens": 131072, + "max_tokens": 131072, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, "tensormesh/Qwen/Qwen3.5-397B-A17B-FP8": { "litellm_provider": "tensormesh", "mode": "chat", diff --git a/provider_endpoints_support.json b/provider_endpoints_support.json index eb27d3fe810..4e6e35e846e 100644 --- a/provider_endpoints_support.json +++ b/provider_endpoints_support.json @@ -121,6 +121,24 @@ "interactions": true } }, + "aiand": { + "display_name": "ai& (`aiand`)", + "url": "https://docs.aiand.com/api/chat-completions/", + "endpoints": { + "chat_completions": true, + "messages": true, + "responses": true, + "embeddings": false, + "image_generations": false, + "audio_transcriptions": false, + "audio_speech": false, + "moderations": false, + "batches": false, + "rerank": false, + "a2a": false, + "interactions": false + } + }, "amazon_nova": { "display_name": "Amazon Nova (`amazon_nova`)", "url": "https://docs.litellm.ai/docs/providers/amazon_nova", diff --git a/scripts/sync_aiand_models.py b/scripts/sync_aiand_models.py new file mode 100644 index 00000000000..4cb3798bf20 --- /dev/null +++ b/scripts/sync_aiand_models.py @@ -0,0 +1,310 @@ +"""Sync the aiand entries of model_prices_and_context_window.json with aiand's live model spec. + +Pulls ``GET https://api.aiand.com/v1/api.json`` (public, no auth), maps spec fields onto +registry fields, and diffs the result against the registry. Dry run (the default) prints the +diff summary and the generated PR body; ``--write`` applies the changes to the root cost map +and its ``litellm/`` backup copy. + +Policy highlights: +- Prices arrive per 1M tokens with float artifacts and are normalized to clean per-token values. +- Registry entries are never deleted; a model absent from the live spec is stamped with + ``metadata.absent_from_spec_since`` (a real, PR-worthy file change) and surfaced as a + warning for a human deprecation call; the stamp is cleared when the model reappears in + the spec. +- The spec cannot express endpoint support or caching behavior, so ``supported_endpoints`` and + ``supports_prompt_caching`` stay fixed for the whole provider. +""" + +import argparse +import json +import sys +from collections.abc import Mapping, Sequence +from dataclasses import dataclass +from datetime import datetime, timezone +from pathlib import Path +from typing import Final + +import httpx +from pydantic import BaseModel, TypeAdapter, ValidationError + +SPEC_URL: Final = "https://api.aiand.com/v1/api.json" +PROVIDER: Final = "aiand" +PREFIX: Final = "aiand/" +SOURCE_URL: Final = "https://api.aiand.com/v1/api.json" +SUPPORTED_ENDPOINTS: Final = ("/v1/chat/completions", "/v1/responses", "/v1/messages") +COST_MAP_RELPATHS: Final = ( + "model_prices_and_context_window.json", + "litellm/model_prices_and_context_window_backup.json", +) + + +class SyncError(RuntimeError): + pass + + +class SpecCost(BaseModel): + input: float + output: float + cache_read: float + + +class SpecLimit(BaseModel): + context: int + output: int + + +class SpecModalities(BaseModel): + input: list[str] + + +class SpecModel(BaseModel): + id: str + name: str + family: str + reasoning: bool + tool_call: bool + structured_output: bool + temperature: bool + attachment: bool + open_weights: bool + cost: SpecCost + limit: SpecLimit + modalities: SpecModalities + + +SPEC_ADAPTER: Final = TypeAdapter(dict[str, SpecModel]) + +RegistryEntry = dict[str, object] +CostMap = dict[str, object] + + +@dataclass(frozen=True, slots=True) +class SyncOutcome: + cost_map: CostMap + added: tuple[str, ...] = () + updated: tuple[str, ...] = () + removed: tuple[str, ...] = () + warnings: tuple[str, ...] = () + + @property + def has_changes(self) -> bool: + return bool(self.added or self.updated or self.removed or self.warnings) + + +def per_token(price_per_million: float) -> float: + return float(f"{price_per_million / 1e6:.6g}") + + +def _today() -> str: + return datetime.now(tz=timezone.utc).date().isoformat() + + +def _spec_fields(model: SpecModel) -> RegistryEntry: + return { + "litellm_provider": PROVIDER, + "mode": "chat", + "input_cost_per_token": per_token(model.cost.input), + "output_cost_per_token": per_token(model.cost.output), + "cache_read_input_token_cost": per_token(model.cost.cache_read), + "max_input_tokens": model.limit.context, + "max_output_tokens": model.limit.output, + "max_tokens": model.limit.output, + "supports_function_calling": model.tool_call, + "supports_native_streaming": True, + "supports_parallel_function_calling": model.tool_call, + "supports_tool_choice": model.tool_call, + "supports_response_schema": model.structured_output, + "supports_prompt_caching": True, + "supports_system_messages": True, + "supports_reasoning": model.reasoning, + "supports_vision": "image" in model.modalities.input, + "source": SOURCE_URL, + "supported_endpoints": list(SUPPORTED_ENDPOINTS), + } + + +def _new_entry(model: SpecModel) -> RegistryEntry: + return _spec_fields(model) + + +def _updated_entry(entry: RegistryEntry, model: SpecModel) -> tuple[RegistryEntry, tuple[str, ...]]: + desired: Final = _spec_fields(model) + changes: Final = tuple( + f"{name}: {entry.get(name)!r} -> {value!r}" for name, value in desired.items() if entry.get(name) != value + ) + extras: Final = dict(sorted((name, value) for name, value in entry.items() if name not in desired)) + return {**desired, **extras}, changes + + +def _with_new_keys_in_block(original: CostMap, result: CostMap, new_keys: Sequence[str]) -> CostMap: + provider_keys: Final = tuple(key for key in original if key.startswith(PREFIX)) + if not new_keys or not provider_keys: + return result + block_end: Final = provider_keys[-1] + return { + key: value + for existing in original + for key, value in ( + (existing, result[existing]), + *((new, result[new]) for new in sorted(new_keys) if existing == block_end), + ) + } + + +def compute_sync(cost_map: CostMap, spec: Mapping[str, SpecModel]) -> SyncOutcome: + spec_ids: Final = frozenset(spec) + registry_ids: Final = {key.removeprefix(PREFIX): key for key in cost_map if key.startswith(PREFIX)} + + added: Final[list[str]] = [] + updated: Final[list[str]] = [] + removed: Final[list[str]] = [] + warnings: Final[list[str]] = [] + result: Final[CostMap] = dict(cost_map) + + for model_id, model in sorted(spec.items()): + key: Final = f"{PREFIX}{model_id}" + entry = result.get(key) + if not isinstance(entry, dict): + result[key] = _new_entry(model) + added.append(key) + continue + new_entry, changes = _updated_entry(entry, model) + metadata = new_entry.get("metadata") + stamped = metadata.get("absent_from_spec_since") if isinstance(metadata, dict) else None + if stamped is not None: + remaining_metadata: Final = { + name: value for name, value in metadata.items() if name != "absent_from_spec_since" + } + if remaining_metadata: + new_entry["metadata"] = dict(sorted(remaining_metadata.items())) + else: + new_entry.pop("metadata") + changes = ( + *changes, + f"metadata.absent_from_spec_since: {stamped!r} -> None (model reappeared in the spec)", + ) + if changes: + updated.append(f"{key}: " + "; ".join(changes)) + result[key] = new_entry + + for model_id, key in sorted(registry_ids.items()): + if model_id in spec_ids: + continue + removed.append(key) + warnings.append( + f"`{key}` is absent from the live spec; the registry entry is kept (never deleted) " + "and needs a human deprecation call" + ) + entry = result.get(key) + if not isinstance(entry, dict): + continue + metadata = entry.get("metadata") + curated: Final = metadata.get("absent_from_spec_since") if isinstance(metadata, dict) else None + if curated is not None: + continue + extras: Final = dict(metadata) if isinstance(metadata, dict) else {} + stamped_entry: Final = dict(entry) + stamped_entry["metadata"] = dict(sorted({**extras, "absent_from_spec_since": _today()}.items())) + result[key] = dict(sorted(stamped_entry.items())) + updated.append(f"{key}: metadata.absent_from_spec_since: None -> {_today()!r}") + return SyncOutcome( + cost_map=_with_new_keys_in_block(cost_map, result, tuple(added)), + added=tuple(added), + updated=tuple(updated), + removed=tuple(removed), + warnings=tuple(warnings), + ) + + +def _section_block(title: str, lines: Sequence[str], backtick: bool) -> str: + bullets: Final = "\n".join(f"- `{line}`" if backtick else f"- {line}" for line in lines) or "- none" + return f"### {title} ({len(lines)})\n{bullets}\n" + + +def render_pr_body(outcome: SyncOutcome) -> str: + return ( + "Automated daily sync of the aiand entries in model_prices_and_context_window.json against " + f"`GET {SPEC_URL}` by scripts/sync_aiand_models.py.\n" + "\n" + f"{_section_block('Added', outcome.added, backtick=True)}" + "\n" + f"{_section_block('Updated', outcome.updated, backtick=True)}" + "\n" + f"{_section_block('Removed from the spec', outcome.warnings, backtick=False)}" + ) + + +def render_summary(outcome: SyncOutcome) -> str: + return ( + f"added={len(outcome.added)} updated={len(outcome.updated)} " + f"removed={len(outcome.removed)} warnings={len(outcome.warnings)}" + ) + + +def load_spec(raw: bytes) -> dict[str, SpecModel]: + parsed: Final = json.loads(raw) + provider: Final = parsed.get("aiand") if isinstance(parsed, dict) else None + models: Final = provider.get("models") if isinstance(provider, dict) else None + try: + spec: Final = SPEC_ADAPTER.validate_python(models) + except ValidationError as error: + raise SyncError(f"the spec response no longer matches the expected shape: {error}") from error + if not spec: + raise SyncError("the spec response contains no aiand models; refusing to rewrite the registry") + for model_id, model in spec.items(): + if model.id != model_id: + raise SyncError(f"spec model id {model.id!r} does not match its key {model_id!r}") + return spec + + +def _fetch(url: str) -> bytes: + response: Final = httpx.get(url, timeout=30, follow_redirects=True) + if response.status_code != 200: + raise SyncError(f"GET {url} returned {response.status_code}") + return response.content + + +def _serialize(cost_map: CostMap) -> str: + return json.dumps(cost_map, indent=4, ensure_ascii=False) + "\n" + + +def main(argv: Sequence[str]) -> int: + parser: Final = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--write", action="store_true", help="apply the sync to the cost map files (default: dry run)") + parser.add_argument("--spec-json", type=Path, help="recorded spec response to use instead of the live API") + parser.add_argument("--pr-body-file", type=Path, help="write the generated PR body to this path") + parser.add_argument("--repo-root", type=Path, default=Path(__file__).resolve().parent.parent) + args: Final = parser.parse_args(argv) + + if args.spec_json is not None: + spec_raw: Final = args.spec_json.read_bytes() + else: + spec_raw = _fetch(SPEC_URL) # rebind-ok: branch-dependent source + spec: Final = load_spec(spec_raw) + + cost_map_path: Final = args.repo_root / COST_MAP_RELPATHS[0] + cost_map: Final = json.loads(cost_map_path.read_text()) + outcome: Final = compute_sync(cost_map, spec) + body: Final = render_pr_body(outcome) + + if args.pr_body_file is not None and outcome.has_changes: + args.pr_body_file.write_text(body) + if args.write and (outcome.added or outcome.updated): + for relpath in COST_MAP_RELPATHS: + (args.repo_root / relpath).write_text(_serialize(outcome.cost_map)) + print(render_summary(outcome)) + print() + print(body) + if not args.write: + print("dry run: no files were touched") + elif not (outcome.added or outcome.updated): + print("registry already in sync: no files were touched") + return 0 + + +if __name__ == "__main__": + try: + raise SystemExit(main(sys.argv[1:])) + except SyncError as error: + print(f"SYNC FAILED: {error}", file=sys.stderr) + raise SystemExit(1) from error diff --git a/tests/unit/llms/openai_like/test_aiand_provider.py b/tests/unit/llms/openai_like/test_aiand_provider.py new file mode 100644 index 00000000000..c21bbbb5bc4 --- /dev/null +++ b/tests/unit/llms/openai_like/test_aiand_provider.py @@ -0,0 +1,250 @@ +""" +Tests for aiand provider configuration and integration. +""" + +import json +from pathlib import Path +from typing import Final + +import pytest +import respx + +import litellm +from litellm.caching.llm_caching_handler import LLMClientCache + + +def test_aiand_provider_resolution(monkeypatch: pytest.MonkeyPatch) -> None: + from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider + + monkeypatch.setenv("AIAND_API_KEY", "aiand-test-key") + + model, provider, api_key, api_base = get_llm_provider( + model="aiand/deepseek-ai/deepseek-v4.1-flash", + custom_llm_provider=None, + api_base=None, + api_key=None, + ) + + assert model == "deepseek-ai/deepseek-v4.1-flash" + assert provider == "aiand" + assert api_key == "aiand-test-key" + assert api_base == "https://api.aiand.com/v1" + + +def test_aiand_provider_keeps_explicit_credentials(monkeypatch: pytest.MonkeyPatch) -> None: + from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider + + monkeypatch.setenv("AIAND_API_KEY", "aiand-env-key") + + _, provider, api_key, api_base = get_llm_provider( + model="aiand/deepseek-ai/deepseek-v4.1-flash", + custom_llm_provider=None, + api_base="https://aiand.internal.example/v1", + api_key="aiand-explicit-key", + ) + + assert provider == "aiand" + assert api_key == "aiand-explicit-key" + assert api_base == "https://aiand.internal.example/v1" + + +def test_aiand_url_autodetection(monkeypatch: pytest.MonkeyPatch) -> None: + from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider + + monkeypatch.setenv("AIAND_API_KEY", "aiand-test-key") + + model, provider, api_key, api_base = get_llm_provider( + model="deepseek-v4.1-flash", + custom_llm_provider=None, + api_base="https://api.aiand.com/v1", + api_key=None, + ) + + assert model == "deepseek-v4.1-flash" + assert provider == "aiand" + assert api_key == "aiand-test-key" + assert api_base == "https://api.aiand.com/v1" + + +AIAND_MODELS = tuple(sorted(name for name in litellm.model_cost if name.startswith("aiand/"))) + + +@pytest.mark.parametrize("model", AIAND_MODELS) +def test_aiand_model_cost_and_capabilities(model: str) -> None: + from litellm.cost_calculator import cost_per_token + + prompt_cost, completion_cost = cost_per_token( + model=model, + prompt_tokens=1_000_000, + completion_tokens=1_000_000, + custom_llm_provider="aiand", + ) + model_info = litellm.get_model_info(model) + + assert prompt_cost == pytest.approx(model_info["input_cost_per_token"] * 1_000_000) + assert completion_cost == pytest.approx(model_info["output_cost_per_token"] * 1_000_000) + assert 0 < model_info["cache_read_input_token_cost"] < model_info["input_cost_per_token"] + assert model_info["output_cost_per_token"] > 0 + assert model_info["max_tokens"] == model_info["max_output_tokens"] <= model_info["max_input_tokens"] + assert model_info["litellm_provider"] == "aiand" + assert model_info["mode"] == "chat" + assert type(model_info["supports_function_calling"]) is bool + assert type(model_info["supports_native_streaming"]) is bool + assert type(model_info["supports_reasoning"]) is bool + assert type(model_info["supports_response_schema"]) is bool + assert litellm.supports_vision(model) is model_info["supports_vision"] + + +def test_aiand_backup_registry_mirrors_cost_map() -> None: + package_root = Path(litellm.__file__).parent + cost_map = json.loads((package_root.parent / "model_prices_and_context_window.json").read_text()) + backup = json.loads((package_root / "model_prices_and_context_window_backup.json").read_text()) + aiand_entries = {name: entry for name, entry in cost_map.items() if name.startswith("aiand/")} + + assert tuple(sorted(aiand_entries)) == AIAND_MODELS + assert aiand_entries + assert all("supports_vision" in entry for entry in aiand_entries.values()) + assert aiand_entries == {name: backup[name] for name in aiand_entries} + + +def test_aiand_is_available_in_add_model_form() -> None: + fields_path = Path(litellm.__file__).parent / "proxy" / "public_endpoints" / "provider_create_fields.json" + providers = json.loads(fields_path.read_text()) + aiand = next(provider for provider in providers if provider["litellm_provider"] == "aiand") + + assert aiand["provider"] == "AIAND" + assert aiand["provider_display_name"] == "ai&" + assert aiand["default_model_placeholder"] == "aiand/deepseek-ai/deepseek-v4.1-flash" + assert {field["key"]: field["required"] for field in aiand["credential_fields"]} == { + "api_base": False, + "api_key": True, + } + + +def test_aiand_supported_endpoints() -> None: + matrix_path = Path(litellm.__file__).parent / "provider_endpoints_support_backup.json" + providers = json.loads(matrix_path.read_text())["providers"] + + assert providers["aiand"]["endpoints"] == { + "chat_completions": True, + "messages": True, + "responses": True, + "embeddings": False, + "image_generations": False, + "audio_transcriptions": False, + "audio_speech": False, + "moderations": False, + "batches": False, + "rerank": False, + "a2a": False, + "interactions": False, + } + + +def test_aiand_chat_completion_request() -> None: + with respx.mock() as upstream: + route: Final = upstream.post("https://api.aiand.com/v1/chat/completions").respond( + 200, + json={ + "id": "chatcmpl_aiand", + "object": "chat.completion", + "created": 1_789_550_000, + "model": "deepseek-ai/deepseek-v4.1-flash", + "choices": [ + { + "index": 0, + "message": {"role": "assistant", "content": "Hello from aiand"}, + "finish_reason": "stop", + } + ], + "usage": {"prompt_tokens": 4, "completion_tokens": 3, "total_tokens": 7}, + }, + ) + response: Final = litellm.completion( + model="aiand/deepseek-ai/deepseek-v4.1-flash", + messages=[{"role": "user", "content": "Say hello"}], + api_key="aiand-test-key", + ) + + request: Final = route.calls.last.request + body: Final = json.loads(request.content) + assert route.call_count == 1 + assert str(request.url) == "https://api.aiand.com/v1/chat/completions" + assert request.headers["authorization"] == "Bearer aiand-test-key" + assert body["model"] == "deepseek-ai/deepseek-v4.1-flash" + assert body["messages"] == [{"role": "user", "content": "Say hello"}] + assert response.choices[0].message.content == "Hello from aiand" + + +def test_aiand_responses_request() -> None: + with respx.mock() as upstream: + route: Final = upstream.post("https://api.aiand.com/v1/responses").respond( + 200, + json={ + "id": "resp_aiand", + "object": "response", + "created_at": 1_789_550_000, + "model": "deepseek-ai/deepseek-v4.1-flash", + "status": "completed", + "output": [ + { + "id": "msg_aiand", + "type": "message", + "role": "assistant", + "status": "completed", + "content": [{"type": "output_text", "text": "Hello from aiand", "annotations": []}], + } + ], + "usage": {"input_tokens": 4, "output_tokens": 3, "total_tokens": 7}, + }, + ) + response: Final = litellm.responses( + model="aiand/deepseek-ai/deepseek-v4.1-flash", + input="Say hello", + api_key="aiand-test-key", + ) + + request: Final = route.calls.last.request + body: Final = json.loads(request.content) + assert route.call_count == 1 + assert str(request.url) == "https://api.aiand.com/v1/responses" + assert request.headers["authorization"] == "Bearer aiand-test-key" + assert body["model"] == "deepseek-ai/deepseek-v4.1-flash" + assert body["input"] == "Say hello" + assert response.output[0].content[0].text == "Hello from aiand" + + +@pytest.mark.asyncio +async def test_aiand_anthropic_messages_request(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setattr(litellm, "disable_aiohttp_transport", True) + monkeypatch.setattr(litellm, "in_memory_llm_clients_cache", LLMClientCache()) + with respx.mock() as upstream: + route: Final = upstream.post("https://api.aiand.com/v1/messages").respond( + 200, + json={ + "id": "msg_aiand", + "type": "message", + "role": "assistant", + "model": "deepseek-ai/deepseek-v4.1-flash", + "content": [{"type": "text", "text": "Hello from aiand"}], + "stop_reason": "end_turn", + "stop_sequence": None, + "usage": {"input_tokens": 4, "output_tokens": 3}, + }, + ) + response: Final = await litellm.anthropic.messages.acreate( + model="aiand/deepseek-ai/deepseek-v4.1-flash", + messages=[{"role": "user", "content": "Say hello"}], + max_tokens=32, + api_key="aiand-test-key", + ) + + request: Final = route.calls.last.request + body: Final = json.loads(request.content) + assert route.call_count == 1 + assert str(request.url) == "https://api.aiand.com/v1/messages" + assert request.headers["authorization"] == "Bearer aiand-test-key" + assert request.headers["anthropic-version"] == "2023-06-01" + assert body["model"] == "deepseek-ai/deepseek-v4.1-flash" + assert body["messages"] == [{"role": "user", "content": "Say hello"}] + assert response["content"][0]["text"] == "Hello from aiand" diff --git a/tests/unit/test_sync_aiand_models.py b/tests/unit/test_sync_aiand_models.py new file mode 100644 index 00000000000..a4bde9071da --- /dev/null +++ b/tests/unit/test_sync_aiand_models.py @@ -0,0 +1,256 @@ +import importlib.util +import json +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[2] +SCRIPT = ROOT / "scripts" / "sync_aiand_models.py" + +_spec = importlib.util.spec_from_file_location("sync_aiand_models", SCRIPT) +assert _spec is not None and _spec.loader is not None +sync = importlib.util.module_from_spec(_spec) +_spec.loader.exec_module(sync) + + +def _model(**overrides: object) -> dict[str, object]: + model: dict[str, object] = { + "id": "acme/chat-1", + "name": "Chat 1", + "family": "chat", + "reasoning": False, + "tool_call": True, + "structured_output": True, + "temperature": True, + "attachment": False, + "open_weights": False, + "cost": {"input": 1.0, "output": 2.0, "cache_read": 0.5}, + "limit": {"context": 8192, "output": 1024}, + "modalities": {"input": ["text"]}, + } + model.update(overrides) + return model + + +def _spec_json(*models: dict[str, object]) -> bytes: + payload: dict[str, object] = {"aiand": {"models": {model["id"]: model for model in models}}} + return json.dumps(payload).encode() + + +@pytest.mark.parametrize( + ("per_million", "expected"), + [ + (3, 3e-06), + (15, 1.5e-05), + (1.4, 1.4e-06), + (0.25999999999999995, 2.6e-07), + (0.060000000000000005, 6e-08), + (1.0399999999999998, 1.04e-06), + (0, 0.0), + ], +) +def test_per_token_normalizes_float_artifacts(per_million: float, expected: float) -> None: + assert sync.per_token(per_million) == expected + + +def test_load_spec_raises_on_shape_change() -> None: + with pytest.raises(sync.SyncError): + sync.load_spec(b'{"aiand": {"models": [{"id": "x"}]}}') + + +def test_load_spec_raises_when_no_models_remain() -> None: + with pytest.raises(sync.SyncError): + sync.load_spec(b'{"aiand": {"models": {}}}') + + +def test_load_spec_raises_when_id_mismatches_key() -> None: + raw = json.dumps({"aiand": {"models": {"acme/chat-1": _model(id="acme/other")}}}).encode() + with pytest.raises(sync.SyncError): + sync.load_spec(raw) + + +def test_added_model_lands_in_cost_map_with_expected_fields() -> None: + spec = sync.load_spec(_spec_json(_model())) + outcome = sync.compute_sync({}, spec) + entry = outcome.cost_map["aiand/acme/chat-1"] + assert entry["litellm_provider"] == "aiand" + assert entry["mode"] == "chat" + assert entry["input_cost_per_token"] == 1e-06 + assert entry["output_cost_per_token"] == 2e-06 + assert entry["cache_read_input_token_cost"] == 5e-07 + assert entry["max_input_tokens"] == 8192 + assert entry["max_output_tokens"] == 1024 + assert entry["max_tokens"] == 1024 + assert entry["supports_function_calling"] is True + assert entry["supports_parallel_function_calling"] is True + assert entry["supports_tool_choice"] is True + assert entry["supports_response_schema"] is True + assert entry["supports_reasoning"] is False + assert entry["supports_vision"] is False + assert entry["source"] == "https://api.aiand.com/v1/api.json" + assert entry["supported_endpoints"] == ["/v1/chat/completions", "/v1/responses", "/v1/messages"] + assert outcome.added == ("aiand/acme/chat-1",) + assert outcome.has_changes is True + + +def test_updated_price_is_detected_and_rendered() -> None: + baseline = sync.compute_sync({}, sync.load_spec(_spec_json(_model()))) + changed = _model(cost={"input": 3.0, "output": 2.0, "cache_read": 0.5}) + outcome = sync.compute_sync(baseline.cost_map, sync.load_spec(_spec_json(changed))) + assert outcome.added == () + assert len(outcome.updated) == 1 + assert outcome.updated[0].startswith("aiand/acme/chat-1:") + assert "input_cost_per_token" in outcome.updated[0] + assert outcome.cost_map["aiand/acme/chat-1"]["input_cost_per_token"] == 3e-06 + assert outcome.has_changes is True + + +def test_removed_model_is_stamped_and_counted_as_updated( + monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setattr(sync, "_today", lambda: "2026-10-03") + baseline = sync.compute_sync({}, sync.load_spec(_spec_json(_model(), _model(id="acme/chat-2", name="Chat 2")))) + remaining = sync.load_spec(_spec_json(_model())) + outcome = sync.compute_sync(baseline.cost_map, remaining) + assert outcome.added == () + assert outcome.removed == ("aiand/acme/chat-2",) + assert len(outcome.updated) == 1 + assert outcome.updated[0].startswith("aiand/acme/chat-2:") + assert "absent_from_spec_since" in outcome.updated[0] + assert outcome.cost_map["aiand/acme/chat-2"]["metadata"] == {"absent_from_spec_since": "2026-10-03"} + assert any("aiand/acme/chat-2" in warning and "human" in warning for warning in outcome.warnings) + assert outcome.has_changes is True + + +def test_already_stamped_absent_model_keeps_the_earliest_date( + monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setattr(sync, "_today", lambda: "2026-10-03") + baseline = sync.compute_sync({}, sync.load_spec(_spec_json(_model(), _model(id="acme/chat-2", name="Chat 2")))) + stamped_entry = dict(baseline.cost_map["aiand/acme/chat-2"]) + stamped_entry["metadata"] = {"absent_from_spec_since": "2026-09-01"} + registry = {**baseline.cost_map, "aiand/acme/chat-2": stamped_entry} + outcome = sync.compute_sync(registry, sync.load_spec(_spec_json(_model()))) + assert outcome.removed == ("aiand/acme/chat-2",) + assert outcome.updated == () + assert outcome.cost_map["aiand/acme/chat-2"]["metadata"] == {"absent_from_spec_since": "2026-09-01"} + assert outcome.has_changes is True + + +def test_reappeared_model_clears_the_stamp_and_counts_as_updated() -> None: + spec = sync.load_spec(_spec_json(_model())) + baseline = sync.compute_sync({}, spec).cost_map + stamped_entry = dict(baseline["aiand/acme/chat-1"]) + stamped_entry["metadata"] = {"absent_from_spec_since": "2026-09-01"} + registry = {**baseline, "aiand/acme/chat-1": stamped_entry} + outcome = sync.compute_sync(registry, spec) + assert outcome.added == () + assert outcome.removed == () + assert len(outcome.updated) == 1 + assert outcome.updated[0].startswith("aiand/acme/chat-1:") + assert "absent_from_spec_since" in outcome.updated[0] + assert "metadata" not in outcome.cost_map["aiand/acme/chat-1"] + assert outcome.has_changes is True + + +def test_updated_entry_preserves_the_absence_marker_as_an_extra() -> None: + model = sync.load_spec(_spec_json(_model()))["acme/chat-1"] + entry = {**sync._new_entry(model), "metadata": {"absent_from_spec_since": "2026-09-01"}} + new_entry, changes = sync._updated_entry(entry, model) + assert new_entry["metadata"] == {"absent_from_spec_since": "2026-09-01"} + assert changes == () + + +def test_parallel_function_calling_follows_tool_call() -> None: + spec = sync.load_spec(_spec_json(_model(tool_call=False))) + outcome = sync.compute_sync({}, spec) + entry = outcome.cost_map["aiand/acme/chat-1"] + assert entry["supports_function_calling"] is False + assert entry["supports_parallel_function_calling"] is False + assert entry["supports_tool_choice"] is False + + +def test_new_keys_land_at_the_end_of_the_provider_block() -> None: + registry = { + "aaa": {}, + "aiand/acme/chat-1": {}, + "zzz": {}, + } + spec = sync.load_spec(_spec_json(_model(), _model(id="acme/new", name="New"))) + outcome = sync.compute_sync(registry, spec) + assert list(outcome.cost_map) == ["aaa", "aiand/acme/chat-1", "aiand/acme/new", "zzz"] + + +def test_pr_body_renders_none_placeholders_for_empty_sections() -> None: + outcome = sync.compute_sync({}, sync.load_spec(_spec_json(_model()))) + body = sync.render_pr_body(outcome) + assert "### Added (1)" in body + assert "### Updated (0)\n- none" in body + assert "### Removed from the spec (0)\n- none" in body + assert sync.render_summary(outcome) == "added=1 updated=0 removed=0 warnings=0" + + +def test_write_updates_root_and_backup_maps_identically(tmp_path: Path) -> None: + repo_root = tmp_path + for relpath in sync.COST_MAP_RELPATHS: + target = repo_root / relpath + target.parent.mkdir(parents=True, exist_ok=True) + target.write_text("{}\n") + spec_path = repo_root / "spec.json" + spec_path.write_bytes(_spec_json(_model())) + pr_body = repo_root / "pr_body.md" + exit_code = sync.main( + [ + "--write", + "--spec-json", + str(spec_path), + "--pr-body-file", + str(pr_body), + "--repo-root", + str(repo_root), + ] + ) + assert exit_code == 0 + root_map = json.loads((repo_root / sync.COST_MAP_RELPATHS[0]).read_text()) + backup_map = json.loads((repo_root / sync.COST_MAP_RELPATHS[1]).read_text()) + assert root_map == backup_map + assert "aiand/acme/chat-1" in root_map + + +def test_removal_only_sync_stamps_and_writes_files( + tmp_path: Path, capsys: pytest.CaptureFixture[str], monkeypatch: pytest.MonkeyPatch +) -> None: + monkeypatch.setattr(sync, "_today", lambda: "2026-10-03") + repo_root = tmp_path + spec = sync.load_spec(_spec_json(_model(), _model(id="acme/chat-2", name="Chat 2"))) + registry = sync.compute_sync({}, spec).cost_map + for relpath in sync.COST_MAP_RELPATHS: + target = repo_root / relpath + target.parent.mkdir(parents=True, exist_ok=True) + target.write_text(json.dumps(registry, indent=4) + "\n") + remaining = repo_root / "remaining.json" + remaining.write_bytes(_spec_json(_model())) + pr_body = repo_root / "pr_body.md" + exit_code = sync.main( + [ + "--write", + "--spec-json", + str(remaining), + "--pr-body-file", + str(pr_body), + "--repo-root", + str(repo_root), + ] + ) + assert exit_code == 0 + assert "updated=1 removed=1 warnings=1" in capsys.readouterr().out + body = pr_body.read_text() + assert "### Added (0)" in body + assert "### Updated (1)" in body + assert "### Removed from the spec (1)" in body + assert "`aiand/acme/chat-2`" in body + assert "### Warnings needing a human call" not in body + root_map = json.loads((repo_root / sync.COST_MAP_RELPATHS[0]).read_text()) + backup_map = json.loads((repo_root / sync.COST_MAP_RELPATHS[1]).read_text()) + assert root_map == backup_map + assert root_map["aiand/acme/chat-2"]["metadata"] == {"absent_from_spec_since": "2026-10-03"} diff --git a/ui/litellm-dashboard/public/assets/logos/aiand.svg b/ui/litellm-dashboard/public/assets/logos/aiand.svg new file mode 100644 index 00000000000..da00acc02ac --- /dev/null +++ b/ui/litellm-dashboard/public/assets/logos/aiand.svg @@ -0,0 +1 @@ +ai&ai& diff --git a/ui/litellm-dashboard/src/components/provider_info_helpers.test.tsx b/ui/litellm-dashboard/src/components/provider_info_helpers.test.tsx index 7cfdaf3275d..91d14ff20e6 100644 --- a/ui/litellm-dashboard/src/components/provider_info_helpers.test.tsx +++ b/ui/litellm-dashboard/src/components/provider_info_helpers.test.tsx @@ -73,6 +73,19 @@ describe("provider_info_helpers", () => { expect(fromEnumKey.logo).toBe(providerLogoMap[Providers.SCX_AI]); }); + it("should map aiand slug to the ai& display name and logo", () => { + const fromSlug = getProviderLogoAndName("aiand"); + expect(fromSlug.displayName).toBe(Providers.AIAND); + expect(fromSlug.logo).toBe(providerLogoMap[Providers.AIAND]); + expect(fromSlug.logo).toBeTruthy(); + }); + + it("should map AIAND enum key to the ai& display name and logo", () => { + const fromEnumKey = getProviderLogoAndName("AIAND"); + expect(fromEnumKey.displayName).toBe(Providers.AIAND); + expect(fromEnumKey.logo).toBe(providerLogoMap[Providers.AIAND]); + }); + it("should map bedrock_mantle slug to Bedrock Mantle display name and logo", () => { const result = getProviderLogoAndName("bedrock_mantle"); expect(result.displayName).toBe(Providers.BedrockMantle); @@ -229,6 +242,10 @@ describe("provider_info_helpers", () => { expect(getPlaceholder(Providers.SCX_AI)).toBe("scx-ai/GLM-5.2"); }); + it("should return an aiand model placeholder for AIAND provider", () => { + expect(getPlaceholder(Providers.AIAND)).toBe("aiand/deepseek-ai/deepseek-v4.1-flash"); + }); + it("should return an edenai model placeholder for EDENAI provider", () => { expect(getPlaceholder(Providers.EDENAI)).toBe("edenai/openai/gpt-mini-latest"); }); diff --git a/ui/litellm-dashboard/src/components/provider_info_helpers.tsx b/ui/litellm-dashboard/src/components/provider_info_helpers.tsx index 5ea693bea10..dcb450b355f 100644 --- a/ui/litellm-dashboard/src/components/provider_info_helpers.tsx +++ b/ui/litellm-dashboard/src/components/provider_info_helpers.tsx @@ -53,6 +53,7 @@ import runwayLogo from "../../public/assets/logos/runway.png"; import sambanovaLogo from "../../public/assets/logos/sambanova.svg"; import sapLogo from "../../public/assets/logos/sap.png"; import scxAiLogo from "../../public/assets/logos/scx_ai.svg"; +import aiandLogo from "../../public/assets/logos/aiand.svg"; import snowflakeLogo from "../../public/assets/logos/snowflake.svg"; import sonioxLogo from "../../public/assets/logos/soniox.svg"; import tencentLogo from "../../public/assets/logos/tencent.svg"; @@ -165,6 +166,7 @@ export enum Providers { Sambanova = "Sambanova", SAP = "SAP Generative AI Hub", SCX_AI = "SCX.ai", + AIAND = "ai&", Snowflake = "Snowflake", Soniox = "Soniox", TEXT_COMPLETION_CODESTRAL = "Text-Completion-Codestral", @@ -285,6 +287,7 @@ export const provider_map: Record = { Sambanova: "sambanova", SAP: "sap", SCX_AI: "scx-ai", + AIAND: "aiand", Snowflake: "snowflake", Soniox: "soniox", TEXT_COMPLETION_CODESTRAL: "text-completion-codestral", @@ -385,6 +388,7 @@ export const providerLogoMap: Partial> = { [Providers.Sambanova]: sambanovaLogo.src, [Providers.SAP]: sapLogo.src, [Providers.SCX_AI]: scxAiLogo.src, + [Providers.AIAND]: aiandLogo.src, [Providers.Snowflake]: snowflakeLogo.src, [Providers.Soniox]: sonioxLogo.src, [Providers.Tencent]: tencentLogo.src, @@ -456,6 +460,7 @@ const providerPlaceholderMap: Partial> = { [Providers.SageMaker]: "sagemaker/jumpstart-dft-meta-textgeneration-llama-2-7b", [Providers.Sail]: "sail/openai/gpt-oss-120b", [Providers.SCX_AI]: "scx-ai/GLM-5.2", + [Providers.AIAND]: "aiand/deepseek-ai/deepseek-v4.1-flash", [Providers.Snowflake]: "snowflake/mistral-7b", [Providers.Tencent]: "tencent/deepseek-v4-pro", [Providers.Vertex_AI]: "gemini-pro", From c77aa2f882b2d059f4d973c7bd551ae18548de4f Mon Sep 17 00:00:00 2001 From: fenil modi Date: Sat, 3 Oct 2026 13:40:17 +0000 Subject: [PATCH 2/4] feat(aiand): reach provider parity with together_ai - declare per-model reasoning_effort_levels on all 13 catalog entries (root + backup cost maps) and map spec reasoning_options in the sync script so re-runs keep the field - register aiand in models_by_provider so get_valid_models, proxy wildcard aiand/*, and cheapest-model picking behave like together_ai - map aiand error payloads (402 insufficient_credits, 401 invalid key, 404 model_not_found) to precise exceptions, including the openai-SDK unwrapped body shape - register /v1/completions so text_completion(model='aiand/...') routes natively with AIAND_API_KEY auth --- litellm/__init__.py | 5 + litellm/constants.py | 1 + .../exception_mapping_utils.py | 77 +++++- litellm/llms/openai_like/providers.json | 2 +- ...odel_prices_and_context_window_backup.json | 61 +++++ model_prices_and_context_window.json | 61 +++++ scripts/sync_aiand_models.py | 20 +- .../test_exception_mapping_utils.py | 236 ++++++++++++++++++ .../llms/openai_like/test_aiand_provider.py | 80 +++++- tests/unit/test_sync_aiand_models.py | 3 +- 10 files changed, 541 insertions(+), 5 deletions(-) diff --git a/litellm/__init__.py b/litellm/__init__.py index b0761da7f7c..27cb7991ef7 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -686,6 +686,7 @@ qwencloud_models: Set = set() qwen_ai_platform_models: Set = set() moonshot_models: Set = set() publicai_models: Set = set() +aiand_models: Set = set() darkbloom_models: Set = set() v0_models: Set = set() morph_models: Set = set() @@ -950,6 +951,8 @@ def _populate_provider_model_sets(model_cost_map: Dict) -> None: moonshot_models.add(key) elif value.get("litellm_provider") == "publicai": publicai_models.add(key) + elif value.get("litellm_provider") == "aiand": + aiand_models.add(key) elif value.get("litellm_provider") == "darkbloom": darkbloom_models.add(key) elif value.get("litellm_provider") == "v0": @@ -1114,6 +1117,7 @@ model_list = list( | qwen_ai_platform_models | moonshot_models | publicai_models + | aiand_models | darkbloom_models | v0_models | morph_models @@ -1226,6 +1230,7 @@ def _build_models_by_provider() -> dict: "modelscope": modelscope_models, "moonshot": moonshot_models, "publicai": publicai_models, + "aiand": aiand_models, "darkbloom": darkbloom_models, "v0": v0_models, "morph": morph_models, diff --git a/litellm/constants.py b/litellm/constants.py index cd40ee22e57..0ced4cfb353 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -1070,6 +1070,7 @@ openai_text_completion_compatible_providers: Final[list] = [ # providers that s "lambda_ai", "hyperbolic", "wandb", + "aiand", ] _openai_like_providers: Final[list] = [ "predibase", diff --git a/litellm/litellm_core_utils/exception_mapping_utils.py b/litellm/litellm_core_utils/exception_mapping_utils.py index 0fdfb301291..5999bb0617d 100644 --- a/litellm/litellm_core_utils/exception_mapping_utils.py +++ b/litellm/litellm_core_utils/exception_mapping_utils.py @@ -1803,6 +1803,71 @@ def _map_together_ai_exception( ) +_AIAND_ERROR_PAYLOAD_KEYS: Final = ("message", "code", "type", "param") + + +def _map_aiand_exception( + *, + model: str, + original_exception: _ProviderHTTPException, + custom_llm_provider: str, + error_str: str, + exception_type: str, + exception_provider: str, + extra_information: str, +) -> None: + error_body: object = getattr(original_exception, "body", None) + if not isinstance(error_body, Mapping): + try: + error_body = json.loads(error_str) + except ValueError: + error_body = None + inner: Final[object] = error_body.get("error") if isinstance(error_body, Mapping) else None + error_payload: Final[Mapping[str, object]] = ( + inner + if isinstance(inner, Mapping) + else error_body + if isinstance(error_body, Mapping) and any(key in error_body for key in _AIAND_ERROR_PAYLOAD_KEYS) + else {} + ) + status_code: Final[int | None] = getattr(original_exception, "status_code", None) + error_message: Final[object] = error_payload.get("message") + message: Final[str] = error_message if isinstance(error_message, str) else error_str + if status_code == 402 and ( + error_payload.get("code") == "insufficient_credits" + or error_payload.get("type") == "billing_error" + or "insufficient_credits" in error_str + or "billing_error" in error_str + ): + raise RateLimitError( + message=f"{exception_provider} - {message}", + llm_provider="aiand", + model=model, + response=getattr(original_exception, "response", None), + litellm_debug_info=extra_information, + ) + elif status_code == 401 and (error_payload.get("code") == "invalid_api_key" or "invalid_api_key" in error_str): + raise AuthenticationError( + message=f"{exception_provider} - {message}", + llm_provider="aiand", + model=model, + response=getattr(original_exception, "response", None), + litellm_debug_info=extra_information, + ) + elif status_code == 404 and ( + error_payload.get("code") == "model_not_found" + or error_payload.get("param") == "model" + or "model_not_found" in error_str + ): + raise NotFoundError( + message=f"{exception_provider} - {message}", + model=model, + llm_provider="aiand", + response=getattr(original_exception, "response", None), + litellm_debug_info=extra_information, + ) + + def _map_aleph_alpha_exception( *, model: str, @@ -2463,7 +2528,17 @@ def exception_type( custom_llm_provider=custom_llm_provider, body=getattr(original_exception, "body", None), ) - if ( + if custom_llm_provider == "aiand": + _map_aiand_exception( + model=model, + original_exception=mappable_exception, + custom_llm_provider=custom_llm_provider, + error_str=error_str, + exception_type=exception_type, + exception_provider=exception_provider, + extra_information=extra_information, + ) + elif ( custom_llm_provider == "openai" or custom_llm_provider == "text-completion-openai" or custom_llm_provider == "custom_openai" diff --git a/litellm/llms/openai_like/providers.json b/litellm/llms/openai_like/providers.json index f0d8f3828ec..847ab731c68 100644 --- a/litellm/llms/openai_like/providers.json +++ b/litellm/llms/openai_like/providers.json @@ -111,7 +111,7 @@ "base_url": "https://api.aiand.com/v1", "api_key_env": "AIAND_API_KEY", "api_base_env": "AIAND_API_BASE", - "supported_endpoints": ["/v1/chat/completions", "/v1/responses", "/v1/messages"] + "supported_endpoints": ["/v1/chat/completions", "/v1/responses", "/v1/messages", "/v1/completions"] }, "crusoe": { "base_url": "https://managed-inference-api-proxy.crusoecloud.com/v1", diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index dcd7477fa7d..a84397093a8 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -59672,6 +59672,11 @@ "/v1/chat/completions", "/v1/responses", "/v1/messages" + ], + "reasoning_effort_levels": [ + "none", + "high", + "max" ] }, "aiand/deepseek-ai/deepseek-v4-pro": { @@ -59697,6 +59702,11 @@ "/v1/chat/completions", "/v1/responses", "/v1/messages" + ], + "reasoning_effort_levels": [ + "none", + "high", + "max" ] }, "aiand/deepseek-ai/deepseek-v4.1-flash": { @@ -59722,6 +59732,11 @@ "/v1/chat/completions", "/v1/responses", "/v1/messages" + ], + "reasoning_effort_levels": [ + "none", + "high", + "max" ] }, "aiand/google/gemma-4-31b-it": { @@ -59747,6 +59762,10 @@ "/v1/chat/completions", "/v1/responses", "/v1/messages" + ], + "reasoning_effort_levels": [ + "none", + "high" ] }, "aiand/moonshotai/kimi-k2.7-code": { @@ -59772,6 +59791,9 @@ "/v1/chat/completions", "/v1/responses", "/v1/messages" + ], + "reasoning_effort_levels": [ + "high" ] }, "aiand/moonshotai/kimi-k3": { @@ -59797,6 +59819,11 @@ "/v1/chat/completions", "/v1/responses", "/v1/messages" + ], + "reasoning_effort_levels": [ + "low", + "high", + "max" ] }, "aiand/motif-technologies/motif-3": { @@ -59822,6 +59849,10 @@ "/v1/chat/completions", "/v1/responses", "/v1/messages" + ], + "reasoning_effort_levels": [ + "none", + "high" ] }, "aiand/openai/gpt-oss-120b": { @@ -59847,6 +59878,11 @@ "/v1/chat/completions", "/v1/responses", "/v1/messages" + ], + "reasoning_effort_levels": [ + "low", + "medium", + "high" ] }, "aiand/qwen/qwen3.6-27b": { @@ -59872,6 +59908,10 @@ "/v1/chat/completions", "/v1/responses", "/v1/messages" + ], + "reasoning_effort_levels": [ + "none", + "high" ] }, "aiand/qwen/qwen3.8-27b": { @@ -59897,6 +59937,12 @@ "/v1/chat/completions", "/v1/responses", "/v1/messages" + ], + "reasoning_effort_levels": [ + "none", + "low", + "medium", + "xhigh" ] }, "aiand/zai-org/glm-5.2": { @@ -59922,6 +59968,11 @@ "/v1/chat/completions", "/v1/responses", "/v1/messages" + ], + "reasoning_effort_levels": [ + "none", + "high", + "max" ] }, "aiand/zai-org/glm-5.3": { @@ -59947,6 +59998,11 @@ "/v1/chat/completions", "/v1/responses", "/v1/messages" + ], + "reasoning_effort_levels": [ + "low", + "high", + "max" ] }, "aiand/zai-org/glm-5.3-flash": { @@ -59972,6 +60028,11 @@ "/v1/chat/completions", "/v1/responses", "/v1/messages" + ], + "reasoning_effort_levels": [ + "low", + "high", + "max" ] }, "tensormesh/Qwen/Qwen3.5-397B-A17B-FP8": { diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index dcd7477fa7d..a84397093a8 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -59672,6 +59672,11 @@ "/v1/chat/completions", "/v1/responses", "/v1/messages" + ], + "reasoning_effort_levels": [ + "none", + "high", + "max" ] }, "aiand/deepseek-ai/deepseek-v4-pro": { @@ -59697,6 +59702,11 @@ "/v1/chat/completions", "/v1/responses", "/v1/messages" + ], + "reasoning_effort_levels": [ + "none", + "high", + "max" ] }, "aiand/deepseek-ai/deepseek-v4.1-flash": { @@ -59722,6 +59732,11 @@ "/v1/chat/completions", "/v1/responses", "/v1/messages" + ], + "reasoning_effort_levels": [ + "none", + "high", + "max" ] }, "aiand/google/gemma-4-31b-it": { @@ -59747,6 +59762,10 @@ "/v1/chat/completions", "/v1/responses", "/v1/messages" + ], + "reasoning_effort_levels": [ + "none", + "high" ] }, "aiand/moonshotai/kimi-k2.7-code": { @@ -59772,6 +59791,9 @@ "/v1/chat/completions", "/v1/responses", "/v1/messages" + ], + "reasoning_effort_levels": [ + "high" ] }, "aiand/moonshotai/kimi-k3": { @@ -59797,6 +59819,11 @@ "/v1/chat/completions", "/v1/responses", "/v1/messages" + ], + "reasoning_effort_levels": [ + "low", + "high", + "max" ] }, "aiand/motif-technologies/motif-3": { @@ -59822,6 +59849,10 @@ "/v1/chat/completions", "/v1/responses", "/v1/messages" + ], + "reasoning_effort_levels": [ + "none", + "high" ] }, "aiand/openai/gpt-oss-120b": { @@ -59847,6 +59878,11 @@ "/v1/chat/completions", "/v1/responses", "/v1/messages" + ], + "reasoning_effort_levels": [ + "low", + "medium", + "high" ] }, "aiand/qwen/qwen3.6-27b": { @@ -59872,6 +59908,10 @@ "/v1/chat/completions", "/v1/responses", "/v1/messages" + ], + "reasoning_effort_levels": [ + "none", + "high" ] }, "aiand/qwen/qwen3.8-27b": { @@ -59897,6 +59937,12 @@ "/v1/chat/completions", "/v1/responses", "/v1/messages" + ], + "reasoning_effort_levels": [ + "none", + "low", + "medium", + "xhigh" ] }, "aiand/zai-org/glm-5.2": { @@ -59922,6 +59968,11 @@ "/v1/chat/completions", "/v1/responses", "/v1/messages" + ], + "reasoning_effort_levels": [ + "none", + "high", + "max" ] }, "aiand/zai-org/glm-5.3": { @@ -59947,6 +59998,11 @@ "/v1/chat/completions", "/v1/responses", "/v1/messages" + ], + "reasoning_effort_levels": [ + "low", + "high", + "max" ] }, "aiand/zai-org/glm-5.3-flash": { @@ -59972,6 +60028,11 @@ "/v1/chat/completions", "/v1/responses", "/v1/messages" + ], + "reasoning_effort_levels": [ + "low", + "high", + "max" ] }, "tensormesh/Qwen/Qwen3.5-397B-A17B-FP8": { diff --git a/scripts/sync_aiand_models.py b/scripts/sync_aiand_models.py index 4cb3798bf20..1b2b97a4752 100644 --- a/scripts/sync_aiand_models.py +++ b/scripts/sync_aiand_models.py @@ -13,6 +13,7 @@ Policy highlights: the spec. - The spec cannot express endpoint support or caching behavior, so ``supported_endpoints`` and ``supports_prompt_caching`` stay fixed for the whole provider. +- Reasoning effort levels map from the spec's ``effort`` reasoning option, when one is declared. """ import argparse @@ -57,11 +58,17 @@ class SpecModalities(BaseModel): input: list[str] +class SpecReasoningOption(BaseModel): + type: str + values: list[str] + + class SpecModel(BaseModel): id: str name: str family: str reasoning: bool + reasoning_options: list[SpecReasoningOption] = [] tool_call: bool structured_output: bool temperature: bool @@ -99,8 +106,16 @@ def _today() -> str: return datetime.now(tz=timezone.utc).date().isoformat() +def _effort_levels(model: SpecModel) -> tuple[str, ...]: + for option in model.reasoning_options: + if option.type == "effort": + return tuple(option.values) + return () + + def _spec_fields(model: SpecModel) -> RegistryEntry: - return { + effort_levels: Final = _effort_levels(model) + fields: RegistryEntry = { "litellm_provider": PROVIDER, "mode": "chat", "input_cost_per_token": per_token(model.cost.input), @@ -121,6 +136,9 @@ def _spec_fields(model: SpecModel) -> RegistryEntry: "source": SOURCE_URL, "supported_endpoints": list(SUPPORTED_ENDPOINTS), } + if effort_levels: + fields["reasoning_effort_levels"] = list(effort_levels) + return fields def _new_entry(model: SpecModel) -> RegistryEntry: diff --git a/tests/unit/litellm_core_utils/test_exception_mapping_utils.py b/tests/unit/litellm_core_utils/test_exception_mapping_utils.py index 9de768ea47b..c1474f3f72b 100644 --- a/tests/unit/litellm_core_utils/test_exception_mapping_utils.py +++ b/tests/unit/litellm_core_utils/test_exception_mapping_utils.py @@ -1,3 +1,5 @@ +import json + import httpx import openai import pytest @@ -9,6 +11,7 @@ from litellm.litellm_core_utils.exception_mapping_utils import ( ExceptionCheckers, _get_body_error_code, _get_response_headers, + _map_aiand_exception, exception_type, extract_and_raise_litellm_exception, ) @@ -1039,6 +1042,239 @@ def test_an_unmapped_exception_with_no_model_or_provider_message_keeps_traceback assert "Traceback (most recent call last)" in raised.value.message +AIAND_INSUFFICIENT_CREDITS_MESSAGE = ( + "Insufficient credits. Review billing at https://console.aiand.com/settings/billing to continue." +) + + +@pytest.mark.parametrize( + "error_body", + [ + { + "message": AIAND_INSUFFICIENT_CREDITS_MESSAGE, + "type": "billing_error", + "param": None, + "code": "insufficient_credits", + }, + { + "message": "Balance too low to process the request.", + "type": "billing_error", + "param": None, + "code": None, + }, + ], +) +def test_an_aiand_402_billing_error_is_a_rate_limit_error(error_body, quiet_exception_mapping): + from litellm.llms.base_llm.chat.transformation import BaseLLMException + + original_exception = BaseLLMException( + status_code=402, + message=json.dumps({"error": error_body}), + ) + + with pytest.raises(litellm.RateLimitError) as raised: + exception_type( + model="test-model", + original_exception=original_exception, + custom_llm_provider="aiand", + ) + + assert raised.value.status_code == 429 + assert raised.value.llm_provider == "aiand" + assert raised.value.model == "test-model" + assert raised.value.message == f"litellm.RateLimitError: AiandException - {error_body['message']}" + + +def test_an_aiand_402_from_the_response_body_is_a_rate_limit_error(quiet_exception_mapping): + from litellm.llms.base_llm.chat.transformation import BaseLLMException + + original_exception = BaseLLMException( + status_code=402, + message=AIAND_INSUFFICIENT_CREDITS_MESSAGE, + body={ + "error": { + "message": AIAND_INSUFFICIENT_CREDITS_MESSAGE, + "type": "billing_error", + "param": None, + "code": "insufficient_credits", + } + }, + ) + + with pytest.raises(litellm.RateLimitError) as raised: + exception_type( + model="test-model", + original_exception=original_exception, + custom_llm_provider="aiand", + ) + + assert raised.value.llm_provider == "aiand" + assert raised.value.message == f"litellm.RateLimitError: AiandException - {AIAND_INSUFFICIENT_CREDITS_MESSAGE}" + + +def test_an_aiand_402_from_an_unwrapped_body_is_a_rate_limit_error(quiet_exception_mapping): + from litellm.llms.base_llm.chat.transformation import BaseLLMException + + original_exception = BaseLLMException( + status_code=402, + message=( + "Error code: 402 - {'error': {'message': " + f"'{AIAND_INSUFFICIENT_CREDITS_MESSAGE}', " + "'type': 'billing_error', 'param': None, 'code': 'insufficient_credits'}}" + ), + body={ + "message": AIAND_INSUFFICIENT_CREDITS_MESSAGE, + "type": "billing_error", + "param": None, + "code": "insufficient_credits", + }, + ) + + with pytest.raises(litellm.RateLimitError) as raised: + exception_type( + model="test-model", + original_exception=original_exception, + custom_llm_provider="aiand", + ) + + assert raised.value.status_code == 429 + assert raised.value.llm_provider == "aiand" + assert raised.value.message == f"litellm.RateLimitError: AiandException - {AIAND_INSUFFICIENT_CREDITS_MESSAGE}" + + +def test_an_aiand_402_from_a_non_json_error_str_is_a_rate_limit_error(quiet_exception_mapping): + from litellm.llms.base_llm.chat.transformation import BaseLLMException + + error_str = ( + "Error code: 402 - {'error': {'message': " + f"'{AIAND_INSUFFICIENT_CREDITS_MESSAGE}', " + "'type': 'billing_error', 'param': None, 'code': 'insufficient_credits'}}" + ) + original_exception = BaseLLMException(status_code=402, message=error_str) + + with pytest.raises(litellm.RateLimitError) as raised: + exception_type( + model="test-model", + original_exception=original_exception, + custom_llm_provider="aiand", + ) + + assert raised.value.status_code == 429 + assert raised.value.llm_provider == "aiand" + assert raised.value.message == f"litellm.RateLimitError: AiandException - {error_str}" + + +def test_an_aiand_401_invalid_api_key_is_an_authentication_error(quiet_exception_mapping): + from litellm.llms.base_llm.chat.transformation import BaseLLMException + + original_exception = BaseLLMException( + status_code=401, + message=json.dumps( + { + "error": { + "message": "Missing or invalid API key", + "type": "authentication_error", + "param": None, + "code": "invalid_api_key", + } + } + ), + ) + + with pytest.raises(litellm.AuthenticationError) as raised: + exception_type( + model="test-model", + original_exception=original_exception, + custom_llm_provider="aiand", + ) + + assert raised.value.status_code == 401 + assert raised.value.llm_provider == "aiand" + assert raised.value.model == "test-model" + assert raised.value.message == "litellm.AuthenticationError: AiandException - Missing or invalid API key" + + +def test_an_aiand_404_model_not_found_is_a_not_found_error(quiet_exception_mapping): + from litellm.llms.base_llm.chat.transformation import BaseLLMException + + original_exception = BaseLLMException( + status_code=404, + message=json.dumps( + { + "error": { + "message": "Model not found", + "type": "invalid_request_error", + "param": "model", + "code": "model_not_found", + } + } + ), + ) + + with pytest.raises(litellm.NotFoundError) as raised: + exception_type( + model="test-model", + original_exception=original_exception, + custom_llm_provider="aiand", + ) + + assert raised.value.status_code == 404 + assert raised.value.llm_provider == "aiand" + assert raised.value.model == "test-model" + assert raised.value.message == "litellm.NotFoundError: AiandException - Model not found" + + +def test_an_unknown_aiand_error_falls_through_without_raising(): + from litellm.llms.base_llm.chat.transformation import BaseLLMException + + class _AiandUpstreamError(BaseLLMException): + code = "" + llm_provider = "aiand" + + original_exception = _AiandUpstreamError(status_code=418, message="I am a teapot") + + assert ( + _map_aiand_exception( + model="test-model", + original_exception=original_exception, + custom_llm_provider="aiand", + error_str="I am a teapot", + exception_type="HTTPException", + exception_provider="AiandException", + extra_information="", + ) + is None + ) + + +def test_an_unknown_aiand_error_still_maps_by_the_upstream_status(quiet_exception_mapping): + from litellm.llms.base_llm.chat.transformation import BaseLLMException + + original_exception = BaseLLMException( + status_code=400, + message=json.dumps( + { + "error": { + "message": "Something else went wrong", + "type": "invalid_request_error", + "param": None, + "code": "invalid_value", + } + } + ), + ) + + with pytest.raises(litellm.BadRequestError) as raised: + exception_type( + model="test-model", + original_exception=original_exception, + custom_llm_provider="aiand", + ) + + assert raised.value.status_code == 400 + assert raised.value.llm_provider == "aiand" + + CONTEXT_WINDOW_MESSAGE = "This model's maximum context length is 4096 tokens." CONTENT_POLICY_MESSAGE = '{"error": {"type": "invalid_request_error", "code": "content_policy_violation"}}' TIMEOUT_MESSAGE = "Request timed out." diff --git a/tests/unit/llms/openai_like/test_aiand_provider.py b/tests/unit/llms/openai_like/test_aiand_provider.py index c21bbbb5bc4..e91391e133f 100644 --- a/tests/unit/llms/openai_like/test_aiand_provider.py +++ b/tests/unit/llms/openai_like/test_aiand_provider.py @@ -4,13 +4,14 @@ Tests for aiand provider configuration and integration. import json from pathlib import Path -from typing import Final +from typing import Final, get_args import pytest import respx import litellm from litellm.caching.llm_caching_handler import LLMClientCache +from litellm.types.llms.openai import REASONING_EFFORT def test_aiand_provider_resolution(monkeypatch: pytest.MonkeyPatch) -> None: @@ -95,6 +96,16 @@ def test_aiand_model_cost_and_capabilities(model: str) -> None: assert litellm.supports_vision(model) is model_info["supports_vision"] +def test_aiand_entries_declare_reasoning_effort_levels() -> None: + known_efforts: Final = frozenset(get_args(REASONING_EFFORT)) + for model in AIAND_MODELS: + levels = litellm.get_model_info(model)["reasoning_effort_levels"] + assert levels, f"{model} declares no reasoning_effort_levels" + assert set(levels) <= known_efforts, f"{model} declares unknown reasoning efforts" + flash = litellm.get_model_info("aiand/deepseek-ai/deepseek-v4.1-flash") + assert set(flash["reasoning_effort_levels"]) == {"none", "high", "max"} + + def test_aiand_backup_registry_mirrors_cost_map() -> None: package_root = Path(litellm.__file__).parent cost_map = json.loads((package_root.parent / "model_prices_and_context_window.json").read_text()) @@ -107,6 +118,22 @@ def test_aiand_backup_registry_mirrors_cost_map() -> None: assert aiand_entries == {name: backup[name] for name in aiand_entries} +def test_aiand_models_listed_by_provider(monkeypatch: pytest.MonkeyPatch) -> None: + package_root = Path(litellm.__file__).parent + backup = json.loads((package_root / "model_prices_and_context_window_backup.json").read_text()) + aiand_keys = {name for name in backup if name.startswith("aiand/")} + assert len(aiand_keys) == 13 + + monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") + monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) + monkeypatch.setattr(litellm, "models_by_provider", dict(litellm.models_by_provider)) + litellm.add_known_models() + + assert "aiand" in litellm.models_by_provider + assert set(litellm.models_by_provider["aiand"]) == aiand_keys + assert set(litellm.get_valid_models(custom_llm_provider="aiand")) == aiand_keys + + def test_aiand_is_available_in_add_model_form() -> None: fields_path = Path(litellm.__file__).parent / "proxy" / "public_endpoints" / "provider_create_fields.json" providers = json.loads(fields_path.read_text()) @@ -141,6 +168,19 @@ def test_aiand_supported_endpoints() -> None: } +def test_aiand_registered_for_text_completion() -> None: + assert "aiand" in litellm.openai_text_completion_compatible_providers + + +def test_aiand_provider_declares_completions_endpoint() -> None: + from litellm.llms.openai_like.json_loader import JSONProviderRegistry + + provider_config: Final = JSONProviderRegistry.get("aiand") + + assert provider_config is not None + assert "/v1/completions" in provider_config.supported_endpoints + + def test_aiand_chat_completion_request() -> None: with respx.mock() as upstream: route: Final = upstream.post("https://api.aiand.com/v1/chat/completions").respond( @@ -214,6 +254,44 @@ def test_aiand_responses_request() -> None: assert response.output[0].content[0].text == "Hello from aiand" +def test_aiand_text_completion_request(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("AIAND_API_KEY", "aiand-test-key") + + with respx.mock() as upstream: + route: Final = upstream.post("https://api.aiand.com/v1/completions").respond( + 200, + json={ + "id": "cmpl_aiand", + "object": "text_completion", + "created": 1_789_550_000, + "model": "deepseek-ai/deepseek-v4.1-flash", + "choices": [ + { + "text": "Hello from aiand", + "index": 0, + "logprobs": None, + "finish_reason": "stop", + } + ], + "usage": {"prompt_tokens": 2, "completion_tokens": 4, "total_tokens": 6}, + }, + ) + response: Final = litellm.text_completion( + model="aiand/deepseek-ai/deepseek-v4.1-flash", + prompt="Say hello", + ) + + request: Final = route.calls.last.request + body: Final = json.loads(request.content) + assert route.call_count == 1 + assert str(request.url) == "https://api.aiand.com/v1/completions" + assert request.headers["authorization"] == "Bearer aiand-test-key" + assert body["model"] == "deepseek-ai/deepseek-v4.1-flash" + assert body["prompt"] == "Say hello" + assert response.object == "text_completion" + assert response.choices[0].text == "Hello from aiand" + + @pytest.mark.asyncio async def test_aiand_anthropic_messages_request(monkeypatch: pytest.MonkeyPatch) -> None: monkeypatch.setattr(litellm, "disable_aiohttp_transport", True) diff --git a/tests/unit/test_sync_aiand_models.py b/tests/unit/test_sync_aiand_models.py index a4bde9071da..8b5e6051dde 100644 --- a/tests/unit/test_sync_aiand_models.py +++ b/tests/unit/test_sync_aiand_models.py @@ -70,7 +70,7 @@ def test_load_spec_raises_when_id_mismatches_key() -> None: def test_added_model_lands_in_cost_map_with_expected_fields() -> None: - spec = sync.load_spec(_spec_json(_model())) + spec = sync.load_spec(_spec_json(_model(reasoning_options=[{"type": "effort", "values": ["low", "high", "max"]}]))) outcome = sync.compute_sync({}, spec) entry = outcome.cost_map["aiand/acme/chat-1"] assert entry["litellm_provider"] == "aiand" @@ -86,6 +86,7 @@ def test_added_model_lands_in_cost_map_with_expected_fields() -> None: assert entry["supports_tool_choice"] is True assert entry["supports_response_schema"] is True assert entry["supports_reasoning"] is False + assert entry["reasoning_effort_levels"] == ["low", "high", "max"] assert entry["supports_vision"] is False assert entry["source"] == "https://api.aiand.com/v1/api.json" assert entry["supported_endpoints"] == ["/v1/chat/completions", "/v1/responses", "/v1/messages"] From bb22beacec990965d908e716448718ca2e3a411b Mon Sep 17 00:00:00 2001 From: fenil modi Date: Sat, 3 Oct 2026 14:14:12 +0000 Subject: [PATCH 3/4] fix(aiand): address review findings on provider parity - aiand errors that match none of the aiand codes now fall through to the openai-compatible mapper, so vLLM context-window strings keep their ContextWindowExceededError classification instead of a generic BadRequestError - 402 insufficient_credits maps to PermissionDeniedError instead of RateLimitError, so the router no longer retries an exhausted balance - sync script treats reasoning_effort_levels as spec-owned in both directions: a catalog-dropped effort option proposes [] instead of preserving stale levels - tests assert litellm-owned invariants (enum membership, reasoning consistency, backup/runtime parity) instead of pinning vendor facts like exact effort levels or a 13-model count --- .../exception_mapping_utils.py | 6 +-- scripts/sync_aiand_models.py | 3 +- .../test_exception_mapping_utils.py | 53 +++++++++++++------ .../llms/openai_like/test_aiand_provider.py | 14 ++--- tests/unit/test_sync_aiand_models.py | 12 +++++ 5 files changed, 62 insertions(+), 26 deletions(-) diff --git a/litellm/litellm_core_utils/exception_mapping_utils.py b/litellm/litellm_core_utils/exception_mapping_utils.py index 5999bb0617d..a4948b3daa4 100644 --- a/litellm/litellm_core_utils/exception_mapping_utils.py +++ b/litellm/litellm_core_utils/exception_mapping_utils.py @@ -1839,11 +1839,11 @@ def _map_aiand_exception( or "insufficient_credits" in error_str or "billing_error" in error_str ): - raise RateLimitError( + raise PermissionDeniedError( message=f"{exception_provider} - {message}", llm_provider="aiand", model=model, - response=getattr(original_exception, "response", None), + response=_response_or_stub(original_exception, status_code=403), litellm_debug_info=extra_information, ) elif status_code == 401 and (error_payload.get("code") == "invalid_api_key" or "invalid_api_key" in error_str): @@ -2538,7 +2538,7 @@ def exception_type( exception_provider=exception_provider, extra_information=extra_information, ) - elif ( + if ( custom_llm_provider == "openai" or custom_llm_provider == "text-completion-openai" or custom_llm_provider == "custom_openai" diff --git a/scripts/sync_aiand_models.py b/scripts/sync_aiand_models.py index 1b2b97a4752..06e589eeaa5 100644 --- a/scripts/sync_aiand_models.py +++ b/scripts/sync_aiand_models.py @@ -136,8 +136,7 @@ def _spec_fields(model: SpecModel) -> RegistryEntry: "source": SOURCE_URL, "supported_endpoints": list(SUPPORTED_ENDPOINTS), } - if effort_levels: - fields["reasoning_effort_levels"] = list(effort_levels) + fields["reasoning_effort_levels"] = list(effort_levels) return fields diff --git a/tests/unit/litellm_core_utils/test_exception_mapping_utils.py b/tests/unit/litellm_core_utils/test_exception_mapping_utils.py index c1474f3f72b..3f936e3e77c 100644 --- a/tests/unit/litellm_core_utils/test_exception_mapping_utils.py +++ b/tests/unit/litellm_core_utils/test_exception_mapping_utils.py @@ -1064,7 +1064,7 @@ AIAND_INSUFFICIENT_CREDITS_MESSAGE = ( }, ], ) -def test_an_aiand_402_billing_error_is_a_rate_limit_error(error_body, quiet_exception_mapping): +def test_an_aiand_402_billing_error_is_a_permission_denied_error(error_body, quiet_exception_mapping): from litellm.llms.base_llm.chat.transformation import BaseLLMException original_exception = BaseLLMException( @@ -1072,20 +1072,20 @@ def test_an_aiand_402_billing_error_is_a_rate_limit_error(error_body, quiet_exce message=json.dumps({"error": error_body}), ) - with pytest.raises(litellm.RateLimitError) as raised: + with pytest.raises(litellm.PermissionDeniedError) as raised: exception_type( model="test-model", original_exception=original_exception, custom_llm_provider="aiand", ) - assert raised.value.status_code == 429 + assert raised.value.status_code == 402 assert raised.value.llm_provider == "aiand" assert raised.value.model == "test-model" - assert raised.value.message == f"litellm.RateLimitError: AiandException - {error_body['message']}" + assert raised.value.message == f"litellm.PermissionDeniedError: AiandException - {error_body['message']}" -def test_an_aiand_402_from_the_response_body_is_a_rate_limit_error(quiet_exception_mapping): +def test_an_aiand_402_from_the_response_body_is_a_permission_denied_error(quiet_exception_mapping): from litellm.llms.base_llm.chat.transformation import BaseLLMException original_exception = BaseLLMException( @@ -1101,7 +1101,7 @@ def test_an_aiand_402_from_the_response_body_is_a_rate_limit_error(quiet_excepti }, ) - with pytest.raises(litellm.RateLimitError) as raised: + with pytest.raises(litellm.PermissionDeniedError) as raised: exception_type( model="test-model", original_exception=original_exception, @@ -1109,10 +1109,12 @@ def test_an_aiand_402_from_the_response_body_is_a_rate_limit_error(quiet_excepti ) assert raised.value.llm_provider == "aiand" - assert raised.value.message == f"litellm.RateLimitError: AiandException - {AIAND_INSUFFICIENT_CREDITS_MESSAGE}" + assert ( + raised.value.message == f"litellm.PermissionDeniedError: AiandException - {AIAND_INSUFFICIENT_CREDITS_MESSAGE}" + ) -def test_an_aiand_402_from_an_unwrapped_body_is_a_rate_limit_error(quiet_exception_mapping): +def test_an_aiand_402_from_an_unwrapped_body_is_a_permission_denied_error(quiet_exception_mapping): from litellm.llms.base_llm.chat.transformation import BaseLLMException original_exception = BaseLLMException( @@ -1130,19 +1132,21 @@ def test_an_aiand_402_from_an_unwrapped_body_is_a_rate_limit_error(quiet_excepti }, ) - with pytest.raises(litellm.RateLimitError) as raised: + with pytest.raises(litellm.PermissionDeniedError) as raised: exception_type( model="test-model", original_exception=original_exception, custom_llm_provider="aiand", ) - assert raised.value.status_code == 429 + assert raised.value.status_code == 402 assert raised.value.llm_provider == "aiand" - assert raised.value.message == f"litellm.RateLimitError: AiandException - {AIAND_INSUFFICIENT_CREDITS_MESSAGE}" + assert ( + raised.value.message == f"litellm.PermissionDeniedError: AiandException - {AIAND_INSUFFICIENT_CREDITS_MESSAGE}" + ) -def test_an_aiand_402_from_a_non_json_error_str_is_a_rate_limit_error(quiet_exception_mapping): +def test_an_aiand_402_from_a_non_json_error_str_is_a_permission_denied_error(quiet_exception_mapping): from litellm.llms.base_llm.chat.transformation import BaseLLMException error_str = ( @@ -1152,16 +1156,16 @@ def test_an_aiand_402_from_a_non_json_error_str_is_a_rate_limit_error(quiet_exce ) original_exception = BaseLLMException(status_code=402, message=error_str) - with pytest.raises(litellm.RateLimitError) as raised: + with pytest.raises(litellm.PermissionDeniedError) as raised: exception_type( model="test-model", original_exception=original_exception, custom_llm_provider="aiand", ) - assert raised.value.status_code == 429 + assert raised.value.status_code == 402 assert raised.value.llm_provider == "aiand" - assert raised.value.message == f"litellm.RateLimitError: AiandException - {error_str}" + assert raised.value.message == f"litellm.PermissionDeniedError: AiandException - {error_str}" def test_an_aiand_401_invalid_api_key_is_an_authentication_error(quiet_exception_mapping): @@ -1224,6 +1228,25 @@ def test_an_aiand_404_model_not_found_is_a_not_found_error(quiet_exception_mappi assert raised.value.message == "litellm.NotFoundError: AiandException - Model not found" +def test_an_aiand_context_window_error_is_a_context_window_exceeded_error(quiet_exception_mapping): + from litellm.llms.base_llm.chat.transformation import BaseLLMException + + original_exception = BaseLLMException(status_code=400, message=CONTEXT_WINDOW_MESSAGE) + + with pytest.raises(litellm.ContextWindowExceededError) as raised: + exception_type( + model="test-model", + original_exception=original_exception, + custom_llm_provider="aiand", + ) + + assert type(raised.value) is litellm.ContextWindowExceededError + assert raised.value.status_code == 400 + assert raised.value.llm_provider == "aiand" + assert raised.value.model == "test-model" + assert f"ContextWindowExceededError: AiandException - {CONTEXT_WINDOW_MESSAGE}" in raised.value.message + + def test_an_unknown_aiand_error_falls_through_without_raising(): from litellm.llms.base_llm.chat.transformation import BaseLLMException diff --git a/tests/unit/llms/openai_like/test_aiand_provider.py b/tests/unit/llms/openai_like/test_aiand_provider.py index e91391e133f..72637b119ad 100644 --- a/tests/unit/llms/openai_like/test_aiand_provider.py +++ b/tests/unit/llms/openai_like/test_aiand_provider.py @@ -96,14 +96,16 @@ def test_aiand_model_cost_and_capabilities(model: str) -> None: assert litellm.supports_vision(model) is model_info["supports_vision"] -def test_aiand_entries_declare_reasoning_effort_levels() -> None: +def test_aiand_reasoning_effort_levels_are_valid() -> None: known_efforts: Final = frozenset(get_args(REASONING_EFFORT)) for model in AIAND_MODELS: - levels = litellm.get_model_info(model)["reasoning_effort_levels"] - assert levels, f"{model} declares no reasoning_effort_levels" + model_info = litellm.get_model_info(model) + levels = model_info.get("reasoning_effort_levels", []) assert set(levels) <= known_efforts, f"{model} declares unknown reasoning efforts" - flash = litellm.get_model_info("aiand/deepseek-ai/deepseek-v4.1-flash") - assert set(flash["reasoning_effort_levels"]) == {"none", "high", "max"} + if levels: + assert model_info["supports_reasoning"] is True, ( + f"{model} declares reasoning efforts without supports_reasoning" + ) def test_aiand_backup_registry_mirrors_cost_map() -> None: @@ -122,7 +124,7 @@ def test_aiand_models_listed_by_provider(monkeypatch: pytest.MonkeyPatch) -> Non package_root = Path(litellm.__file__).parent backup = json.loads((package_root / "model_prices_and_context_window_backup.json").read_text()) aiand_keys = {name for name in backup if name.startswith("aiand/")} - assert len(aiand_keys) == 13 + assert aiand_keys monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) diff --git a/tests/unit/test_sync_aiand_models.py b/tests/unit/test_sync_aiand_models.py index 8b5e6051dde..d3ae4497d39 100644 --- a/tests/unit/test_sync_aiand_models.py +++ b/tests/unit/test_sync_aiand_models.py @@ -106,6 +106,18 @@ def test_updated_price_is_detected_and_rendered() -> None: assert outcome.has_changes is True +def test_dropped_effort_option_is_reported_and_cleared() -> None: + offered = _model(reasoning_options=[{"type": "effort", "values": ["low", "high"]}]) + baseline = sync.compute_sync({}, sync.load_spec(_spec_json(offered))) + outcome = sync.compute_sync(baseline.cost_map, sync.load_spec(_spec_json(_model()))) + assert outcome.added == () + assert len(outcome.updated) == 1 + assert outcome.updated[0].startswith("aiand/acme/chat-1:") + assert "reasoning_effort_levels" in outcome.updated[0] + assert outcome.cost_map["aiand/acme/chat-1"]["reasoning_effort_levels"] == [] + assert outcome.has_changes is True + + def test_removed_model_is_stamped_and_counted_as_updated( monkeypatch: pytest.MonkeyPatch, ) -> None: From ff632d868ce5fb46c9baccc5a80af2d745a28cd9 Mon Sep 17 00:00:00 2001 From: fenil modi Date: Sat, 3 Oct 2026 15:22:06 +0000 Subject: [PATCH 4/4] fix(aiand): stay within the basedpyright reportUnknownArgumentType ceiling - derive exception_provider as the AiandException literal at the aiand dispatch (the branch is unreachable for any other provider, so the value is unchanged) and narrow model with an isinstance Final, so the call stops feeding legacy untyped locals into the typed mapper - annotate aiand_models as Set[str] --- litellm/__init__.py | 2 +- litellm/litellm_core_utils/exception_mapping_utils.py | 5 +++-- 2 files changed, 4 insertions(+), 3 deletions(-) diff --git a/litellm/__init__.py b/litellm/__init__.py index 27cb7991ef7..6fb2680ec9c 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -686,7 +686,7 @@ qwencloud_models: Set = set() qwen_ai_platform_models: Set = set() moonshot_models: Set = set() publicai_models: Set = set() -aiand_models: Set = set() +aiand_models: Set[str] = set() darkbloom_models: Set = set() v0_models: Set = set() morph_models: Set = set() diff --git a/litellm/litellm_core_utils/exception_mapping_utils.py b/litellm/litellm_core_utils/exception_mapping_utils.py index a4948b3daa4..db8d4b81d40 100644 --- a/litellm/litellm_core_utils/exception_mapping_utils.py +++ b/litellm/litellm_core_utils/exception_mapping_utils.py @@ -2529,13 +2529,14 @@ def exception_type( body=getattr(original_exception, "body", None), ) if custom_llm_provider == "aiand": + _aiand_model: Final = model if isinstance(model, str) else "" _map_aiand_exception( - model=model, + model=_aiand_model, original_exception=mappable_exception, custom_llm_provider=custom_llm_provider, error_str=error_str, exception_type=exception_type, - exception_provider=exception_provider, + exception_provider="AiandException", extra_information=extra_information, ) if (