diff --git a/.github/workflows/sync-aiand-models.yml b/.github/workflows/sync-aiand-models.yml new file mode 100644 index 00000000000..b476844b8d1 --- /dev/null +++ b/.github/workflows/sync-aiand-models.yml @@ -0,0 +1,73 @@ +name: Sync aiand model registry + +on: + schedule: + - cron: "45 7 * * *" + workflow_dispatch: + +concurrency: + group: sync-aiand-models + cancel-in-progress: false + +permissions: + contents: write + pull-requests: write + +jobs: + sync_aiand_models: + if: github.repository == 'BerriAI/litellm' + runs-on: ubuntu-latest + env: + BASE_BRANCH: ${{ github.event.repository.default_branch }} + steps: + - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + ref: ${{ env.BASE_BRANCH }} + persist-credentials: false + - name: Set up uv + uses: ./.github/actions/setup-uv-with-retries + with: + version: "0.10.9" + - name: Look for an already-open sync PR + id: existing + run: | + open_pr="$(gh pr list --repo "$GITHUB_REPOSITORY" --state open --limit 1000 --json headRefName,changedFiles \ + --jq '[.[] | select(.headRefName | startswith("litellm_aiand_registry_sync_")) | select(.changedFiles > 0)] | first | .headRefName // empty')" + echo "open_pr=$open_pr" >> "$GITHUB_OUTPUT" + if [ -n "$open_pr" ]; then + echo "Sync PR $open_pr still has unreviewed changes; skipping this run." + fi + env: + GH_TOKEN: ${{ secrets.GH_TOKEN || github.token }} + - name: Run the sync + if: steps.existing.outputs.open_pr == '' + run: | + uv run --frozen python scripts/sync_aiand_models.py --write --pr-body-file "$RUNNER_TEMP/pr_body.md" + - name: Regenerate the JSON schema + if: steps.existing.outputs.open_pr == '' + run: | + uv run --frozen python ci_cd/generate_model_prices_schema.py + - name: Create a pull request when the registry changed + if: steps.existing.outputs.open_pr == '' + run: | + if git diff --quiet; then + echo "Registry already in sync; no PR needed." + exit 0 + fi + branch="litellm_aiand_registry_sync_$(date +'%Y-%m-%d')" + git config user.name "github-actions[bot]" + git config user.email "41898282+github-actions[bot]@users.noreply.github.com" + git checkout -b "$branch" + git add model_prices_and_context_window.json \ + litellm/model_prices_and_context_window_backup.json \ + model_prices_and_context_window.schema.json + git commit -m "feat(models): sync aiand model registry $(date +'%Y-%m-%d')" + gh auth setup-git + git push origin --delete "$branch" 2>/dev/null || true + git push origin "$branch" + gh pr create --title "feat(models): sync aiand model registry" \ + --body-file "$RUNNER_TEMP/pr_body.md" \ + --head "$branch" \ + --base "$BASE_BRANCH" + env: + GH_TOKEN: ${{ secrets.GH_TOKEN || github.token }} diff --git a/README.md b/README.md index 4004e6474ee..a0b85c0a4b0 100644 --- a/README.md +++ b/README.md @@ -298,6 +298,7 @@ Set `LITELLM_PROXY_API_BASE` and `LITELLM_PROXY_API_KEY` and every model call th | Provider | `/chat/completions` | `/messages` | `/responses` | `/embeddings` | `/image/generations` | `/audio/transcriptions` | `/audio/speech` | `/moderations` | `/batches` | `/rerank` | |-------------------------------------------------------------------------------------|---------------------|-------------|--------------|---------------|----------------------|-------------------------|-----------------|----------------|-----------|-----------| | [Abliteration (`abliteration`)](https://docs.litellm.ai/docs/providers/abliteration) | ✅ | | | | | | | | | | +| [ai& (`aiand`)](https://docs.aiand.com/) | ✅ | ✅ | ✅ | | | | | | | | | [AI/ML API (`aiml`)](https://docs.litellm.ai/docs/providers/aiml) | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | | | [AI21 (`ai21`)](https://docs.litellm.ai/docs/providers/ai21) | ✅ | ✅ | ✅ | | | | | | | | | [AI21 Chat (`ai21_chat`)](https://docs.litellm.ai/docs/providers/ai21) | ✅ | ✅ | ✅ | | | | | | | | diff --git a/litellm/constants.py b/litellm/constants.py index 18fc6aa7e74..cd40ee22e57 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -939,6 +939,7 @@ openai_compatible_endpoints: Final[list] = [ "https://dashscope.aliyuncs.com/compatible-mode/v1", "https://api-inference.modelscope.cn/v1", "https://api.moonshot.ai/v1", + "https://api.aiand.com/v1", "https://api.publicai.co/v1", "https://api.synthetic.new/openai/v1", "https://serverless.tensormesh.ai/v1", @@ -1001,6 +1002,7 @@ openai_compatible_providers: Final[list] = [ "chatgpt", # ChatGPT subscription API "novita", "meta_llama", + "aiand", "publicai", # PublicAI - JSON-configured provider "synthetic", # Synthetic - JSON-configured provider "tensormesh", # Tensormesh - JSON-configured provider diff --git a/litellm/llms/openai_like/providers.json b/litellm/llms/openai_like/providers.json index 61ff4be3a46..f0d8f3828ec 100644 --- a/litellm/llms/openai_like/providers.json +++ b/litellm/llms/openai_like/providers.json @@ -107,6 +107,12 @@ "api_key_env": "AIHUBMIX_API_KEY", "api_base_env": "AIHUBMIX_API_BASE" }, + "aiand": { + "base_url": "https://api.aiand.com/v1", + "api_key_env": "AIAND_API_KEY", + "api_base_env": "AIAND_API_BASE", + "supported_endpoints": ["/v1/chat/completions", "/v1/responses", "/v1/messages"] + }, "crusoe": { "base_url": "https://managed-inference-api-proxy.crusoecloud.com/v1", "api_key_env": "CRUSOE_API_KEY", diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 7e8f9bb4093..dcd7477fa7d 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -59649,6 +59649,331 @@ ], "supports_audio_input": true }, + "aiand/deepseek-ai/deepseek-v4-flash": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 2.5e-07, + "cache_read_input_token_cost": 8e-08, + "max_input_tokens": 1048576, + "max_output_tokens": 384000, + "max_tokens": 384000, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/deepseek-ai/deepseek-v4-pro": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1e-06, + "output_cost_per_token": 2.5e-06, + "cache_read_input_token_cost": 2.5e-07, + "max_input_tokens": 1048576, + "max_output_tokens": 384000, + "max_tokens": 384000, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/deepseek-ai/deepseek-v4.1-flash": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 3e-07, + "output_cost_per_token": 6e-07, + "cache_read_input_token_cost": 2e-08, + "max_input_tokens": 1048576, + "max_output_tokens": 384000, + "max_tokens": 384000, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/google/gemma-4-31b-it": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 2e-07, + "output_cost_per_token": 5e-07, + "cache_read_input_token_cost": 5e-08, + "max_input_tokens": 262144, + "max_output_tokens": 32768, + "max_tokens": 32768, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/moonshotai/kimi-k2.7-code": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 7.5e-07, + "output_cost_per_token": 3.5e-06, + "cache_read_input_token_cost": 2e-07, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "max_tokens": 262144, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/moonshotai/kimi-k3": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 3e-06, + "output_cost_per_token": 1.25e-05, + "cache_read_input_token_cost": 5e-07, + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "max_tokens": 131072, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/motif-technologies/motif-3": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 5e-07, + "output_cost_per_token": 2e-06, + "cache_read_input_token_cost": 2e-07, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "max_tokens": 262144, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": false, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/openai/gpt-oss-120b": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 6e-07, + "cache_read_input_token_cost": 8e-08, + "max_input_tokens": 131072, + "max_output_tokens": 32768, + "max_tokens": 32768, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/qwen/qwen3.6-27b": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 3.2e-07, + "output_cost_per_token": 3.2e-06, + "cache_read_input_token_cost": 2e-07, + "max_input_tokens": 262144, + "max_output_tokens": 65536, + "max_tokens": 65536, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/qwen/qwen3.8-27b": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 4e-07, + "output_cost_per_token": 3e-06, + "cache_read_input_token_cost": 2e-07, + "max_input_tokens": 262144, + "max_output_tokens": 32768, + "max_tokens": 32768, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/zai-org/glm-5.2": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1e-06, + "output_cost_per_token": 4e-06, + "cache_read_input_token_cost": 3e-07, + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "max_tokens": 131072, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/zai-org/glm-5.3": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1e-06, + "output_cost_per_token": 4e-06, + "cache_read_input_token_cost": 3e-07, + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "max_tokens": 131072, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/zai-org/glm-5.3-flash": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 5e-07, + "cache_read_input_token_cost": 3e-08, + "max_input_tokens": 1048550, + "max_output_tokens": 131072, + "max_tokens": 131072, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, "tensormesh/Qwen/Qwen3.5-397B-A17B-FP8": { "litellm_provider": "tensormesh", "mode": "chat", diff --git a/litellm/provider_endpoints_support_backup.json b/litellm/provider_endpoints_support_backup.json index c9635587eeb..2694af5ada1 100644 --- a/litellm/provider_endpoints_support_backup.json +++ b/litellm/provider_endpoints_support_backup.json @@ -120,6 +120,24 @@ "interactions": true } }, + "aiand": { + "display_name": "ai& (`aiand`)", + "url": "https://docs.aiand.com/api/chat-completions/", + "endpoints": { + "chat_completions": true, + "messages": true, + "responses": true, + "embeddings": false, + "image_generations": false, + "audio_transcriptions": false, + "audio_speech": false, + "moderations": false, + "batches": false, + "rerank": false, + "a2a": false, + "interactions": false + } + }, "amazon_nova": { "display_name": "Amazon Nova (`amazon_nova`)", "url": "https://docs.litellm.ai/docs/providers/amazon_nova", diff --git a/litellm/proxy/public_endpoints/provider_create_fields.json b/litellm/proxy/public_endpoints/provider_create_fields.json index 6e96d6ad0ec..2a8995d0db1 100644 --- a/litellm/proxy/public_endpoints/provider_create_fields.json +++ b/litellm/proxy/public_endpoints/provider_create_fields.json @@ -1,4 +1,32 @@ [ + { + "provider": "AIAND", + "provider_display_name": "ai&", + "litellm_provider": "aiand", + "credential_fields": [ + { + "key": "api_base", + "label": "API Base", + "placeholder": "https://api.aiand.com/v1", + "tooltip": null, + "required": false, + "field_type": "text", + "options": null, + "default_value": null + }, + { + "key": "api_key", + "label": "API Key", + "placeholder": null, + "tooltip": null, + "required": true, + "field_type": "password", + "options": null, + "default_value": null + } + ], + "default_model_placeholder": "aiand/deepseek-ai/deepseek-v4.1-flash" + }, { "provider": "AIML", "provider_display_name": "AI/ML API", diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 6919fd6fd27..5a117b85c30 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -4153,6 +4153,7 @@ class LlmProviders(str, Enum): PARASAIL = "parasail" XIAOMI_MIMO = "xiaomi_mimo" TENSORMESH = "tensormesh" + AIAND = "aiand" LIBERTAI = "libertai" PINSTRIPES = "pinstripes" COGNITION = "cognition" diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 7e8f9bb4093..dcd7477fa7d 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -59649,6 +59649,331 @@ ], "supports_audio_input": true }, + "aiand/deepseek-ai/deepseek-v4-flash": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 2.5e-07, + "cache_read_input_token_cost": 8e-08, + "max_input_tokens": 1048576, + "max_output_tokens": 384000, + "max_tokens": 384000, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/deepseek-ai/deepseek-v4-pro": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1e-06, + "output_cost_per_token": 2.5e-06, + "cache_read_input_token_cost": 2.5e-07, + "max_input_tokens": 1048576, + "max_output_tokens": 384000, + "max_tokens": 384000, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/deepseek-ai/deepseek-v4.1-flash": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 3e-07, + "output_cost_per_token": 6e-07, + "cache_read_input_token_cost": 2e-08, + "max_input_tokens": 1048576, + "max_output_tokens": 384000, + "max_tokens": 384000, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/google/gemma-4-31b-it": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 2e-07, + "output_cost_per_token": 5e-07, + "cache_read_input_token_cost": 5e-08, + "max_input_tokens": 262144, + "max_output_tokens": 32768, + "max_tokens": 32768, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/moonshotai/kimi-k2.7-code": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 7.5e-07, + "output_cost_per_token": 3.5e-06, + "cache_read_input_token_cost": 2e-07, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "max_tokens": 262144, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/moonshotai/kimi-k3": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 3e-06, + "output_cost_per_token": 1.25e-05, + "cache_read_input_token_cost": 5e-07, + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "max_tokens": 131072, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/motif-technologies/motif-3": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 5e-07, + "output_cost_per_token": 2e-06, + "cache_read_input_token_cost": 2e-07, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "max_tokens": 262144, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": false, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/openai/gpt-oss-120b": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 6e-07, + "cache_read_input_token_cost": 8e-08, + "max_input_tokens": 131072, + "max_output_tokens": 32768, + "max_tokens": 32768, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/qwen/qwen3.6-27b": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 3.2e-07, + "output_cost_per_token": 3.2e-06, + "cache_read_input_token_cost": 2e-07, + "max_input_tokens": 262144, + "max_output_tokens": 65536, + "max_tokens": 65536, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/qwen/qwen3.8-27b": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 4e-07, + "output_cost_per_token": 3e-06, + "cache_read_input_token_cost": 2e-07, + "max_input_tokens": 262144, + "max_output_tokens": 32768, + "max_tokens": 32768, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/zai-org/glm-5.2": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1e-06, + "output_cost_per_token": 4e-06, + "cache_read_input_token_cost": 3e-07, + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "max_tokens": 131072, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/zai-org/glm-5.3": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1e-06, + "output_cost_per_token": 4e-06, + "cache_read_input_token_cost": 3e-07, + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "max_tokens": 131072, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, + "aiand/zai-org/glm-5.3-flash": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 5e-07, + "cache_read_input_token_cost": 3e-08, + "max_input_tokens": 1048550, + "max_output_tokens": 131072, + "max_tokens": 131072, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ] + }, "tensormesh/Qwen/Qwen3.5-397B-A17B-FP8": { "litellm_provider": "tensormesh", "mode": "chat", diff --git a/provider_endpoints_support.json b/provider_endpoints_support.json index eb27d3fe810..4e6e35e846e 100644 --- a/provider_endpoints_support.json +++ b/provider_endpoints_support.json @@ -121,6 +121,24 @@ "interactions": true } }, + "aiand": { + "display_name": "ai& (`aiand`)", + "url": "https://docs.aiand.com/api/chat-completions/", + "endpoints": { + "chat_completions": true, + "messages": true, + "responses": true, + "embeddings": false, + "image_generations": false, + "audio_transcriptions": false, + "audio_speech": false, + "moderations": false, + "batches": false, + "rerank": false, + "a2a": false, + "interactions": false + } + }, "amazon_nova": { "display_name": "Amazon Nova (`amazon_nova`)", "url": "https://docs.litellm.ai/docs/providers/amazon_nova", diff --git a/scripts/sync_aiand_models.py b/scripts/sync_aiand_models.py new file mode 100644 index 00000000000..4cb3798bf20 --- /dev/null +++ b/scripts/sync_aiand_models.py @@ -0,0 +1,310 @@ +"""Sync the aiand entries of model_prices_and_context_window.json with aiand's live model spec. + +Pulls ``GET https://api.aiand.com/v1/api.json`` (public, no auth), maps spec fields onto +registry fields, and diffs the result against the registry. Dry run (the default) prints the +diff summary and the generated PR body; ``--write`` applies the changes to the root cost map +and its ``litellm/`` backup copy. + +Policy highlights: +- Prices arrive per 1M tokens with float artifacts and are normalized to clean per-token values. +- Registry entries are never deleted; a model absent from the live spec is stamped with + ``metadata.absent_from_spec_since`` (a real, PR-worthy file change) and surfaced as a + warning for a human deprecation call; the stamp is cleared when the model reappears in + the spec. +- The spec cannot express endpoint support or caching behavior, so ``supported_endpoints`` and + ``supports_prompt_caching`` stay fixed for the whole provider. +""" + +import argparse +import json +import sys +from collections.abc import Mapping, Sequence +from dataclasses import dataclass +from datetime import datetime, timezone +from pathlib import Path +from typing import Final + +import httpx +from pydantic import BaseModel, TypeAdapter, ValidationError + +SPEC_URL: Final = "https://api.aiand.com/v1/api.json" +PROVIDER: Final = "aiand" +PREFIX: Final = "aiand/" +SOURCE_URL: Final = "https://api.aiand.com/v1/api.json" +SUPPORTED_ENDPOINTS: Final = ("/v1/chat/completions", "/v1/responses", "/v1/messages") +COST_MAP_RELPATHS: Final = ( + "model_prices_and_context_window.json", + "litellm/model_prices_and_context_window_backup.json", +) + + +class SyncError(RuntimeError): + pass + + +class SpecCost(BaseModel): + input: float + output: float + cache_read: float + + +class SpecLimit(BaseModel): + context: int + output: int + + +class SpecModalities(BaseModel): + input: list[str] + + +class SpecModel(BaseModel): + id: str + name: str + family: str + reasoning: bool + tool_call: bool + structured_output: bool + temperature: bool + attachment: bool + open_weights: bool + cost: SpecCost + limit: SpecLimit + modalities: SpecModalities + + +SPEC_ADAPTER: Final = TypeAdapter(dict[str, SpecModel]) + +RegistryEntry = dict[str, object] +CostMap = dict[str, object] + + +@dataclass(frozen=True, slots=True) +class SyncOutcome: + cost_map: CostMap + added: tuple[str, ...] = () + updated: tuple[str, ...] = () + removed: tuple[str, ...] = () + warnings: tuple[str, ...] = () + + @property + def has_changes(self) -> bool: + return bool(self.added or self.updated or self.removed or self.warnings) + + +def per_token(price_per_million: float) -> float: + return float(f"{price_per_million / 1e6:.6g}") + + +def _today() -> str: + return datetime.now(tz=timezone.utc).date().isoformat() + + +def _spec_fields(model: SpecModel) -> RegistryEntry: + return { + "litellm_provider": PROVIDER, + "mode": "chat", + "input_cost_per_token": per_token(model.cost.input), + "output_cost_per_token": per_token(model.cost.output), + "cache_read_input_token_cost": per_token(model.cost.cache_read), + "max_input_tokens": model.limit.context, + "max_output_tokens": model.limit.output, + "max_tokens": model.limit.output, + "supports_function_calling": model.tool_call, + "supports_native_streaming": True, + "supports_parallel_function_calling": model.tool_call, + "supports_tool_choice": model.tool_call, + "supports_response_schema": model.structured_output, + "supports_prompt_caching": True, + "supports_system_messages": True, + "supports_reasoning": model.reasoning, + "supports_vision": "image" in model.modalities.input, + "source": SOURCE_URL, + "supported_endpoints": list(SUPPORTED_ENDPOINTS), + } + + +def _new_entry(model: SpecModel) -> RegistryEntry: + return _spec_fields(model) + + +def _updated_entry(entry: RegistryEntry, model: SpecModel) -> tuple[RegistryEntry, tuple[str, ...]]: + desired: Final = _spec_fields(model) + changes: Final = tuple( + f"{name}: {entry.get(name)!r} -> {value!r}" for name, value in desired.items() if entry.get(name) != value + ) + extras: Final = dict(sorted((name, value) for name, value in entry.items() if name not in desired)) + return {**desired, **extras}, changes + + +def _with_new_keys_in_block(original: CostMap, result: CostMap, new_keys: Sequence[str]) -> CostMap: + provider_keys: Final = tuple(key for key in original if key.startswith(PREFIX)) + if not new_keys or not provider_keys: + return result + block_end: Final = provider_keys[-1] + return { + key: value + for existing in original + for key, value in ( + (existing, result[existing]), + *((new, result[new]) for new in sorted(new_keys) if existing == block_end), + ) + } + + +def compute_sync(cost_map: CostMap, spec: Mapping[str, SpecModel]) -> SyncOutcome: + spec_ids: Final = frozenset(spec) + registry_ids: Final = {key.removeprefix(PREFIX): key for key in cost_map if key.startswith(PREFIX)} + + added: Final[list[str]] = [] + updated: Final[list[str]] = [] + removed: Final[list[str]] = [] + warnings: Final[list[str]] = [] + result: Final[CostMap] = dict(cost_map) + + for model_id, model in sorted(spec.items()): + key: Final = f"{PREFIX}{model_id}" + entry = result.get(key) + if not isinstance(entry, dict): + result[key] = _new_entry(model) + added.append(key) + continue + new_entry, changes = _updated_entry(entry, model) + metadata = new_entry.get("metadata") + stamped = metadata.get("absent_from_spec_since") if isinstance(metadata, dict) else None + if stamped is not None: + remaining_metadata: Final = { + name: value for name, value in metadata.items() if name != "absent_from_spec_since" + } + if remaining_metadata: + new_entry["metadata"] = dict(sorted(remaining_metadata.items())) + else: + new_entry.pop("metadata") + changes = ( + *changes, + f"metadata.absent_from_spec_since: {stamped!r} -> None (model reappeared in the spec)", + ) + if changes: + updated.append(f"{key}: " + "; ".join(changes)) + result[key] = new_entry + + for model_id, key in sorted(registry_ids.items()): + if model_id in spec_ids: + continue + removed.append(key) + warnings.append( + f"`{key}` is absent from the live spec; the registry entry is kept (never deleted) " + "and needs a human deprecation call" + ) + entry = result.get(key) + if not isinstance(entry, dict): + continue + metadata = entry.get("metadata") + curated: Final = metadata.get("absent_from_spec_since") if isinstance(metadata, dict) else None + if curated is not None: + continue + extras: Final = dict(metadata) if isinstance(metadata, dict) else {} + stamped_entry: Final = dict(entry) + stamped_entry["metadata"] = dict(sorted({**extras, "absent_from_spec_since": _today()}.items())) + result[key] = dict(sorted(stamped_entry.items())) + updated.append(f"{key}: metadata.absent_from_spec_since: None -> {_today()!r}") + return SyncOutcome( + cost_map=_with_new_keys_in_block(cost_map, result, tuple(added)), + added=tuple(added), + updated=tuple(updated), + removed=tuple(removed), + warnings=tuple(warnings), + ) + + +def _section_block(title: str, lines: Sequence[str], backtick: bool) -> str: + bullets: Final = "\n".join(f"- `{line}`" if backtick else f"- {line}" for line in lines) or "- none" + return f"### {title} ({len(lines)})\n{bullets}\n" + + +def render_pr_body(outcome: SyncOutcome) -> str: + return ( + "Automated daily sync of the aiand entries in model_prices_and_context_window.json against " + f"`GET {SPEC_URL}` by scripts/sync_aiand_models.py.\n" + "\n" + f"{_section_block('Added', outcome.added, backtick=True)}" + "\n" + f"{_section_block('Updated', outcome.updated, backtick=True)}" + "\n" + f"{_section_block('Removed from the spec', outcome.warnings, backtick=False)}" + ) + + +def render_summary(outcome: SyncOutcome) -> str: + return ( + f"added={len(outcome.added)} updated={len(outcome.updated)} " + f"removed={len(outcome.removed)} warnings={len(outcome.warnings)}" + ) + + +def load_spec(raw: bytes) -> dict[str, SpecModel]: + parsed: Final = json.loads(raw) + provider: Final = parsed.get("aiand") if isinstance(parsed, dict) else None + models: Final = provider.get("models") if isinstance(provider, dict) else None + try: + spec: Final = SPEC_ADAPTER.validate_python(models) + except ValidationError as error: + raise SyncError(f"the spec response no longer matches the expected shape: {error}") from error + if not spec: + raise SyncError("the spec response contains no aiand models; refusing to rewrite the registry") + for model_id, model in spec.items(): + if model.id != model_id: + raise SyncError(f"spec model id {model.id!r} does not match its key {model_id!r}") + return spec + + +def _fetch(url: str) -> bytes: + response: Final = httpx.get(url, timeout=30, follow_redirects=True) + if response.status_code != 200: + raise SyncError(f"GET {url} returned {response.status_code}") + return response.content + + +def _serialize(cost_map: CostMap) -> str: + return json.dumps(cost_map, indent=4, ensure_ascii=False) + "\n" + + +def main(argv: Sequence[str]) -> int: + parser: Final = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--write", action="store_true", help="apply the sync to the cost map files (default: dry run)") + parser.add_argument("--spec-json", type=Path, help="recorded spec response to use instead of the live API") + parser.add_argument("--pr-body-file", type=Path, help="write the generated PR body to this path") + parser.add_argument("--repo-root", type=Path, default=Path(__file__).resolve().parent.parent) + args: Final = parser.parse_args(argv) + + if args.spec_json is not None: + spec_raw: Final = args.spec_json.read_bytes() + else: + spec_raw = _fetch(SPEC_URL) # rebind-ok: branch-dependent source + spec: Final = load_spec(spec_raw) + + cost_map_path: Final = args.repo_root / COST_MAP_RELPATHS[0] + cost_map: Final = json.loads(cost_map_path.read_text()) + outcome: Final = compute_sync(cost_map, spec) + body: Final = render_pr_body(outcome) + + if args.pr_body_file is not None and outcome.has_changes: + args.pr_body_file.write_text(body) + if args.write and (outcome.added or outcome.updated): + for relpath in COST_MAP_RELPATHS: + (args.repo_root / relpath).write_text(_serialize(outcome.cost_map)) + print(render_summary(outcome)) + print() + print(body) + if not args.write: + print("dry run: no files were touched") + elif not (outcome.added or outcome.updated): + print("registry already in sync: no files were touched") + return 0 + + +if __name__ == "__main__": + try: + raise SystemExit(main(sys.argv[1:])) + except SyncError as error: + print(f"SYNC FAILED: {error}", file=sys.stderr) + raise SystemExit(1) from error diff --git a/tests/unit/llms/openai_like/test_aiand_provider.py b/tests/unit/llms/openai_like/test_aiand_provider.py new file mode 100644 index 00000000000..c21bbbb5bc4 --- /dev/null +++ b/tests/unit/llms/openai_like/test_aiand_provider.py @@ -0,0 +1,250 @@ +""" +Tests for aiand provider configuration and integration. +""" + +import json +from pathlib import Path +from typing import Final + +import pytest +import respx + +import litellm +from litellm.caching.llm_caching_handler import LLMClientCache + + +def test_aiand_provider_resolution(monkeypatch: pytest.MonkeyPatch) -> None: + from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider + + monkeypatch.setenv("AIAND_API_KEY", "aiand-test-key") + + model, provider, api_key, api_base = get_llm_provider( + model="aiand/deepseek-ai/deepseek-v4.1-flash", + custom_llm_provider=None, + api_base=None, + api_key=None, + ) + + assert model == "deepseek-ai/deepseek-v4.1-flash" + assert provider == "aiand" + assert api_key == "aiand-test-key" + assert api_base == "https://api.aiand.com/v1" + + +def test_aiand_provider_keeps_explicit_credentials(monkeypatch: pytest.MonkeyPatch) -> None: + from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider + + monkeypatch.setenv("AIAND_API_KEY", "aiand-env-key") + + _, provider, api_key, api_base = get_llm_provider( + model="aiand/deepseek-ai/deepseek-v4.1-flash", + custom_llm_provider=None, + api_base="https://aiand.internal.example/v1", + api_key="aiand-explicit-key", + ) + + assert provider == "aiand" + assert api_key == "aiand-explicit-key" + assert api_base == "https://aiand.internal.example/v1" + + +def test_aiand_url_autodetection(monkeypatch: pytest.MonkeyPatch) -> None: + from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider + + monkeypatch.setenv("AIAND_API_KEY", "aiand-test-key") + + model, provider, api_key, api_base = get_llm_provider( + model="deepseek-v4.1-flash", + custom_llm_provider=None, + api_base="https://api.aiand.com/v1", + api_key=None, + ) + + assert model == "deepseek-v4.1-flash" + assert provider == "aiand" + assert api_key == "aiand-test-key" + assert api_base == "https://api.aiand.com/v1" + + +AIAND_MODELS = tuple(sorted(name for name in litellm.model_cost if name.startswith("aiand/"))) + + +@pytest.mark.parametrize("model", AIAND_MODELS) +def test_aiand_model_cost_and_capabilities(model: str) -> None: + from litellm.cost_calculator import cost_per_token + + prompt_cost, completion_cost = cost_per_token( + model=model, + prompt_tokens=1_000_000, + completion_tokens=1_000_000, + custom_llm_provider="aiand", + ) + model_info = litellm.get_model_info(model) + + assert prompt_cost == pytest.approx(model_info["input_cost_per_token"] * 1_000_000) + assert completion_cost == pytest.approx(model_info["output_cost_per_token"] * 1_000_000) + assert 0 < model_info["cache_read_input_token_cost"] < model_info["input_cost_per_token"] + assert model_info["output_cost_per_token"] > 0 + assert model_info["max_tokens"] == model_info["max_output_tokens"] <= model_info["max_input_tokens"] + assert model_info["litellm_provider"] == "aiand" + assert model_info["mode"] == "chat" + assert type(model_info["supports_function_calling"]) is bool + assert type(model_info["supports_native_streaming"]) is bool + assert type(model_info["supports_reasoning"]) is bool + assert type(model_info["supports_response_schema"]) is bool + assert litellm.supports_vision(model) is model_info["supports_vision"] + + +def test_aiand_backup_registry_mirrors_cost_map() -> None: + package_root = Path(litellm.__file__).parent + cost_map = json.loads((package_root.parent / "model_prices_and_context_window.json").read_text()) + backup = json.loads((package_root / "model_prices_and_context_window_backup.json").read_text()) + aiand_entries = {name: entry for name, entry in cost_map.items() if name.startswith("aiand/")} + + assert tuple(sorted(aiand_entries)) == AIAND_MODELS + assert aiand_entries + assert all("supports_vision" in entry for entry in aiand_entries.values()) + assert aiand_entries == {name: backup[name] for name in aiand_entries} + + +def test_aiand_is_available_in_add_model_form() -> None: + fields_path = Path(litellm.__file__).parent / "proxy" / "public_endpoints" / "provider_create_fields.json" + providers = json.loads(fields_path.read_text()) + aiand = next(provider for provider in providers if provider["litellm_provider"] == "aiand") + + assert aiand["provider"] == "AIAND" + assert aiand["provider_display_name"] == "ai&" + assert aiand["default_model_placeholder"] == "aiand/deepseek-ai/deepseek-v4.1-flash" + assert {field["key"]: field["required"] for field in aiand["credential_fields"]} == { + "api_base": False, + "api_key": True, + } + + +def test_aiand_supported_endpoints() -> None: + matrix_path = Path(litellm.__file__).parent / "provider_endpoints_support_backup.json" + providers = json.loads(matrix_path.read_text())["providers"] + + assert providers["aiand"]["endpoints"] == { + "chat_completions": True, + "messages": True, + "responses": True, + "embeddings": False, + "image_generations": False, + "audio_transcriptions": False, + "audio_speech": False, + "moderations": False, + "batches": False, + "rerank": False, + "a2a": False, + "interactions": False, + } + + +def test_aiand_chat_completion_request() -> None: + with respx.mock() as upstream: + route: Final = upstream.post("https://api.aiand.com/v1/chat/completions").respond( + 200, + json={ + "id": "chatcmpl_aiand", + "object": "chat.completion", + "created": 1_789_550_000, + "model": "deepseek-ai/deepseek-v4.1-flash", + "choices": [ + { + "index": 0, + "message": {"role": "assistant", "content": "Hello from aiand"}, + "finish_reason": "stop", + } + ], + "usage": {"prompt_tokens": 4, "completion_tokens": 3, "total_tokens": 7}, + }, + ) + response: Final = litellm.completion( + model="aiand/deepseek-ai/deepseek-v4.1-flash", + messages=[{"role": "user", "content": "Say hello"}], + api_key="aiand-test-key", + ) + + request: Final = route.calls.last.request + body: Final = json.loads(request.content) + assert route.call_count == 1 + assert str(request.url) == "https://api.aiand.com/v1/chat/completions" + assert request.headers["authorization"] == "Bearer aiand-test-key" + assert body["model"] == "deepseek-ai/deepseek-v4.1-flash" + assert body["messages"] == [{"role": "user", "content": "Say hello"}] + assert response.choices[0].message.content == "Hello from aiand" + + +def test_aiand_responses_request() -> None: + with respx.mock() as upstream: + route: Final = upstream.post("https://api.aiand.com/v1/responses").respond( + 200, + json={ + "id": "resp_aiand", + "object": "response", + "created_at": 1_789_550_000, + "model": "deepseek-ai/deepseek-v4.1-flash", + "status": "completed", + "output": [ + { + "id": "msg_aiand", + "type": "message", + "role": "assistant", + "status": "completed", + "content": [{"type": "output_text", "text": "Hello from aiand", "annotations": []}], + } + ], + "usage": {"input_tokens": 4, "output_tokens": 3, "total_tokens": 7}, + }, + ) + response: Final = litellm.responses( + model="aiand/deepseek-ai/deepseek-v4.1-flash", + input="Say hello", + api_key="aiand-test-key", + ) + + request: Final = route.calls.last.request + body: Final = json.loads(request.content) + assert route.call_count == 1 + assert str(request.url) == "https://api.aiand.com/v1/responses" + assert request.headers["authorization"] == "Bearer aiand-test-key" + assert body["model"] == "deepseek-ai/deepseek-v4.1-flash" + assert body["input"] == "Say hello" + assert response.output[0].content[0].text == "Hello from aiand" + + +@pytest.mark.asyncio +async def test_aiand_anthropic_messages_request(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setattr(litellm, "disable_aiohttp_transport", True) + monkeypatch.setattr(litellm, "in_memory_llm_clients_cache", LLMClientCache()) + with respx.mock() as upstream: + route: Final = upstream.post("https://api.aiand.com/v1/messages").respond( + 200, + json={ + "id": "msg_aiand", + "type": "message", + "role": "assistant", + "model": "deepseek-ai/deepseek-v4.1-flash", + "content": [{"type": "text", "text": "Hello from aiand"}], + "stop_reason": "end_turn", + "stop_sequence": None, + "usage": {"input_tokens": 4, "output_tokens": 3}, + }, + ) + response: Final = await litellm.anthropic.messages.acreate( + model="aiand/deepseek-ai/deepseek-v4.1-flash", + messages=[{"role": "user", "content": "Say hello"}], + max_tokens=32, + api_key="aiand-test-key", + ) + + request: Final = route.calls.last.request + body: Final = json.loads(request.content) + assert route.call_count == 1 + assert str(request.url) == "https://api.aiand.com/v1/messages" + assert request.headers["authorization"] == "Bearer aiand-test-key" + assert request.headers["anthropic-version"] == "2023-06-01" + assert body["model"] == "deepseek-ai/deepseek-v4.1-flash" + assert body["messages"] == [{"role": "user", "content": "Say hello"}] + assert response["content"][0]["text"] == "Hello from aiand" diff --git a/tests/unit/test_sync_aiand_models.py b/tests/unit/test_sync_aiand_models.py new file mode 100644 index 00000000000..a4bde9071da --- /dev/null +++ b/tests/unit/test_sync_aiand_models.py @@ -0,0 +1,256 @@ +import importlib.util +import json +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[2] +SCRIPT = ROOT / "scripts" / "sync_aiand_models.py" + +_spec = importlib.util.spec_from_file_location("sync_aiand_models", SCRIPT) +assert _spec is not None and _spec.loader is not None +sync = importlib.util.module_from_spec(_spec) +_spec.loader.exec_module(sync) + + +def _model(**overrides: object) -> dict[str, object]: + model: dict[str, object] = { + "id": "acme/chat-1", + "name": "Chat 1", + "family": "chat", + "reasoning": False, + "tool_call": True, + "structured_output": True, + "temperature": True, + "attachment": False, + "open_weights": False, + "cost": {"input": 1.0, "output": 2.0, "cache_read": 0.5}, + "limit": {"context": 8192, "output": 1024}, + "modalities": {"input": ["text"]}, + } + model.update(overrides) + return model + + +def _spec_json(*models: dict[str, object]) -> bytes: + payload: dict[str, object] = {"aiand": {"models": {model["id"]: model for model in models}}} + return json.dumps(payload).encode() + + +@pytest.mark.parametrize( + ("per_million", "expected"), + [ + (3, 3e-06), + (15, 1.5e-05), + (1.4, 1.4e-06), + (0.25999999999999995, 2.6e-07), + (0.060000000000000005, 6e-08), + (1.0399999999999998, 1.04e-06), + (0, 0.0), + ], +) +def test_per_token_normalizes_float_artifacts(per_million: float, expected: float) -> None: + assert sync.per_token(per_million) == expected + + +def test_load_spec_raises_on_shape_change() -> None: + with pytest.raises(sync.SyncError): + sync.load_spec(b'{"aiand": {"models": [{"id": "x"}]}}') + + +def test_load_spec_raises_when_no_models_remain() -> None: + with pytest.raises(sync.SyncError): + sync.load_spec(b'{"aiand": {"models": {}}}') + + +def test_load_spec_raises_when_id_mismatches_key() -> None: + raw = json.dumps({"aiand": {"models": {"acme/chat-1": _model(id="acme/other")}}}).encode() + with pytest.raises(sync.SyncError): + sync.load_spec(raw) + + +def test_added_model_lands_in_cost_map_with_expected_fields() -> None: + spec = sync.load_spec(_spec_json(_model())) + outcome = sync.compute_sync({}, spec) + entry = outcome.cost_map["aiand/acme/chat-1"] + assert entry["litellm_provider"] == "aiand" + assert entry["mode"] == "chat" + assert entry["input_cost_per_token"] == 1e-06 + assert entry["output_cost_per_token"] == 2e-06 + assert entry["cache_read_input_token_cost"] == 5e-07 + assert entry["max_input_tokens"] == 8192 + assert entry["max_output_tokens"] == 1024 + assert entry["max_tokens"] == 1024 + assert entry["supports_function_calling"] is True + assert entry["supports_parallel_function_calling"] is True + assert entry["supports_tool_choice"] is True + assert entry["supports_response_schema"] is True + assert entry["supports_reasoning"] is False + assert entry["supports_vision"] is False + assert entry["source"] == "https://api.aiand.com/v1/api.json" + assert entry["supported_endpoints"] == ["/v1/chat/completions", "/v1/responses", "/v1/messages"] + assert outcome.added == ("aiand/acme/chat-1",) + assert outcome.has_changes is True + + +def test_updated_price_is_detected_and_rendered() -> None: + baseline = sync.compute_sync({}, sync.load_spec(_spec_json(_model()))) + changed = _model(cost={"input": 3.0, "output": 2.0, "cache_read": 0.5}) + outcome = sync.compute_sync(baseline.cost_map, sync.load_spec(_spec_json(changed))) + assert outcome.added == () + assert len(outcome.updated) == 1 + assert outcome.updated[0].startswith("aiand/acme/chat-1:") + assert "input_cost_per_token" in outcome.updated[0] + assert outcome.cost_map["aiand/acme/chat-1"]["input_cost_per_token"] == 3e-06 + assert outcome.has_changes is True + + +def test_removed_model_is_stamped_and_counted_as_updated( + monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setattr(sync, "_today", lambda: "2026-10-03") + baseline = sync.compute_sync({}, sync.load_spec(_spec_json(_model(), _model(id="acme/chat-2", name="Chat 2")))) + remaining = sync.load_spec(_spec_json(_model())) + outcome = sync.compute_sync(baseline.cost_map, remaining) + assert outcome.added == () + assert outcome.removed == ("aiand/acme/chat-2",) + assert len(outcome.updated) == 1 + assert outcome.updated[0].startswith("aiand/acme/chat-2:") + assert "absent_from_spec_since" in outcome.updated[0] + assert outcome.cost_map["aiand/acme/chat-2"]["metadata"] == {"absent_from_spec_since": "2026-10-03"} + assert any("aiand/acme/chat-2" in warning and "human" in warning for warning in outcome.warnings) + assert outcome.has_changes is True + + +def test_already_stamped_absent_model_keeps_the_earliest_date( + monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setattr(sync, "_today", lambda: "2026-10-03") + baseline = sync.compute_sync({}, sync.load_spec(_spec_json(_model(), _model(id="acme/chat-2", name="Chat 2")))) + stamped_entry = dict(baseline.cost_map["aiand/acme/chat-2"]) + stamped_entry["metadata"] = {"absent_from_spec_since": "2026-09-01"} + registry = {**baseline.cost_map, "aiand/acme/chat-2": stamped_entry} + outcome = sync.compute_sync(registry, sync.load_spec(_spec_json(_model()))) + assert outcome.removed == ("aiand/acme/chat-2",) + assert outcome.updated == () + assert outcome.cost_map["aiand/acme/chat-2"]["metadata"] == {"absent_from_spec_since": "2026-09-01"} + assert outcome.has_changes is True + + +def test_reappeared_model_clears_the_stamp_and_counts_as_updated() -> None: + spec = sync.load_spec(_spec_json(_model())) + baseline = sync.compute_sync({}, spec).cost_map + stamped_entry = dict(baseline["aiand/acme/chat-1"]) + stamped_entry["metadata"] = {"absent_from_spec_since": "2026-09-01"} + registry = {**baseline, "aiand/acme/chat-1": stamped_entry} + outcome = sync.compute_sync(registry, spec) + assert outcome.added == () + assert outcome.removed == () + assert len(outcome.updated) == 1 + assert outcome.updated[0].startswith("aiand/acme/chat-1:") + assert "absent_from_spec_since" in outcome.updated[0] + assert "metadata" not in outcome.cost_map["aiand/acme/chat-1"] + assert outcome.has_changes is True + + +def test_updated_entry_preserves_the_absence_marker_as_an_extra() -> None: + model = sync.load_spec(_spec_json(_model()))["acme/chat-1"] + entry = {**sync._new_entry(model), "metadata": {"absent_from_spec_since": "2026-09-01"}} + new_entry, changes = sync._updated_entry(entry, model) + assert new_entry["metadata"] == {"absent_from_spec_since": "2026-09-01"} + assert changes == () + + +def test_parallel_function_calling_follows_tool_call() -> None: + spec = sync.load_spec(_spec_json(_model(tool_call=False))) + outcome = sync.compute_sync({}, spec) + entry = outcome.cost_map["aiand/acme/chat-1"] + assert entry["supports_function_calling"] is False + assert entry["supports_parallel_function_calling"] is False + assert entry["supports_tool_choice"] is False + + +def test_new_keys_land_at_the_end_of_the_provider_block() -> None: + registry = { + "aaa": {}, + "aiand/acme/chat-1": {}, + "zzz": {}, + } + spec = sync.load_spec(_spec_json(_model(), _model(id="acme/new", name="New"))) + outcome = sync.compute_sync(registry, spec) + assert list(outcome.cost_map) == ["aaa", "aiand/acme/chat-1", "aiand/acme/new", "zzz"] + + +def test_pr_body_renders_none_placeholders_for_empty_sections() -> None: + outcome = sync.compute_sync({}, sync.load_spec(_spec_json(_model()))) + body = sync.render_pr_body(outcome) + assert "### Added (1)" in body + assert "### Updated (0)\n- none" in body + assert "### Removed from the spec (0)\n- none" in body + assert sync.render_summary(outcome) == "added=1 updated=0 removed=0 warnings=0" + + +def test_write_updates_root_and_backup_maps_identically(tmp_path: Path) -> None: + repo_root = tmp_path + for relpath in sync.COST_MAP_RELPATHS: + target = repo_root / relpath + target.parent.mkdir(parents=True, exist_ok=True) + target.write_text("{}\n") + spec_path = repo_root / "spec.json" + spec_path.write_bytes(_spec_json(_model())) + pr_body = repo_root / "pr_body.md" + exit_code = sync.main( + [ + "--write", + "--spec-json", + str(spec_path), + "--pr-body-file", + str(pr_body), + "--repo-root", + str(repo_root), + ] + ) + assert exit_code == 0 + root_map = json.loads((repo_root / sync.COST_MAP_RELPATHS[0]).read_text()) + backup_map = json.loads((repo_root / sync.COST_MAP_RELPATHS[1]).read_text()) + assert root_map == backup_map + assert "aiand/acme/chat-1" in root_map + + +def test_removal_only_sync_stamps_and_writes_files( + tmp_path: Path, capsys: pytest.CaptureFixture[str], monkeypatch: pytest.MonkeyPatch +) -> None: + monkeypatch.setattr(sync, "_today", lambda: "2026-10-03") + repo_root = tmp_path + spec = sync.load_spec(_spec_json(_model(), _model(id="acme/chat-2", name="Chat 2"))) + registry = sync.compute_sync({}, spec).cost_map + for relpath in sync.COST_MAP_RELPATHS: + target = repo_root / relpath + target.parent.mkdir(parents=True, exist_ok=True) + target.write_text(json.dumps(registry, indent=4) + "\n") + remaining = repo_root / "remaining.json" + remaining.write_bytes(_spec_json(_model())) + pr_body = repo_root / "pr_body.md" + exit_code = sync.main( + [ + "--write", + "--spec-json", + str(remaining), + "--pr-body-file", + str(pr_body), + "--repo-root", + str(repo_root), + ] + ) + assert exit_code == 0 + assert "updated=1 removed=1 warnings=1" in capsys.readouterr().out + body = pr_body.read_text() + assert "### Added (0)" in body + assert "### Updated (1)" in body + assert "### Removed from the spec (1)" in body + assert "`aiand/acme/chat-2`" in body + assert "### Warnings needing a human call" not in body + root_map = json.loads((repo_root / sync.COST_MAP_RELPATHS[0]).read_text()) + backup_map = json.loads((repo_root / sync.COST_MAP_RELPATHS[1]).read_text()) + assert root_map == backup_map + assert root_map["aiand/acme/chat-2"]["metadata"] == {"absent_from_spec_since": "2026-10-03"} diff --git a/ui/litellm-dashboard/public/assets/logos/aiand.svg b/ui/litellm-dashboard/public/assets/logos/aiand.svg new file mode 100644 index 00000000000..da00acc02ac --- /dev/null +++ b/ui/litellm-dashboard/public/assets/logos/aiand.svg @@ -0,0 +1 @@ +ai&ai& diff --git a/ui/litellm-dashboard/src/components/provider_info_helpers.test.tsx b/ui/litellm-dashboard/src/components/provider_info_helpers.test.tsx index 7cfdaf3275d..91d14ff20e6 100644 --- a/ui/litellm-dashboard/src/components/provider_info_helpers.test.tsx +++ b/ui/litellm-dashboard/src/components/provider_info_helpers.test.tsx @@ -73,6 +73,19 @@ describe("provider_info_helpers", () => { expect(fromEnumKey.logo).toBe(providerLogoMap[Providers.SCX_AI]); }); + it("should map aiand slug to the ai& display name and logo", () => { + const fromSlug = getProviderLogoAndName("aiand"); + expect(fromSlug.displayName).toBe(Providers.AIAND); + expect(fromSlug.logo).toBe(providerLogoMap[Providers.AIAND]); + expect(fromSlug.logo).toBeTruthy(); + }); + + it("should map AIAND enum key to the ai& display name and logo", () => { + const fromEnumKey = getProviderLogoAndName("AIAND"); + expect(fromEnumKey.displayName).toBe(Providers.AIAND); + expect(fromEnumKey.logo).toBe(providerLogoMap[Providers.AIAND]); + }); + it("should map bedrock_mantle slug to Bedrock Mantle display name and logo", () => { const result = getProviderLogoAndName("bedrock_mantle"); expect(result.displayName).toBe(Providers.BedrockMantle); @@ -229,6 +242,10 @@ describe("provider_info_helpers", () => { expect(getPlaceholder(Providers.SCX_AI)).toBe("scx-ai/GLM-5.2"); }); + it("should return an aiand model placeholder for AIAND provider", () => { + expect(getPlaceholder(Providers.AIAND)).toBe("aiand/deepseek-ai/deepseek-v4.1-flash"); + }); + it("should return an edenai model placeholder for EDENAI provider", () => { expect(getPlaceholder(Providers.EDENAI)).toBe("edenai/openai/gpt-mini-latest"); }); diff --git a/ui/litellm-dashboard/src/components/provider_info_helpers.tsx b/ui/litellm-dashboard/src/components/provider_info_helpers.tsx index 5ea693bea10..dcb450b355f 100644 --- a/ui/litellm-dashboard/src/components/provider_info_helpers.tsx +++ b/ui/litellm-dashboard/src/components/provider_info_helpers.tsx @@ -53,6 +53,7 @@ import runwayLogo from "../../public/assets/logos/runway.png"; import sambanovaLogo from "../../public/assets/logos/sambanova.svg"; import sapLogo from "../../public/assets/logos/sap.png"; import scxAiLogo from "../../public/assets/logos/scx_ai.svg"; +import aiandLogo from "../../public/assets/logos/aiand.svg"; import snowflakeLogo from "../../public/assets/logos/snowflake.svg"; import sonioxLogo from "../../public/assets/logos/soniox.svg"; import tencentLogo from "../../public/assets/logos/tencent.svg"; @@ -165,6 +166,7 @@ export enum Providers { Sambanova = "Sambanova", SAP = "SAP Generative AI Hub", SCX_AI = "SCX.ai", + AIAND = "ai&", Snowflake = "Snowflake", Soniox = "Soniox", TEXT_COMPLETION_CODESTRAL = "Text-Completion-Codestral", @@ -285,6 +287,7 @@ export const provider_map: Record = { Sambanova: "sambanova", SAP: "sap", SCX_AI: "scx-ai", + AIAND: "aiand", Snowflake: "snowflake", Soniox: "soniox", TEXT_COMPLETION_CODESTRAL: "text-completion-codestral", @@ -385,6 +388,7 @@ export const providerLogoMap: Partial> = { [Providers.Sambanova]: sambanovaLogo.src, [Providers.SAP]: sapLogo.src, [Providers.SCX_AI]: scxAiLogo.src, + [Providers.AIAND]: aiandLogo.src, [Providers.Snowflake]: snowflakeLogo.src, [Providers.Soniox]: sonioxLogo.src, [Providers.Tencent]: tencentLogo.src, @@ -456,6 +460,7 @@ const providerPlaceholderMap: Partial> = { [Providers.SageMaker]: "sagemaker/jumpstart-dft-meta-textgeneration-llama-2-7b", [Providers.Sail]: "sail/openai/gpt-oss-120b", [Providers.SCX_AI]: "scx-ai/GLM-5.2", + [Providers.AIAND]: "aiand/deepseek-ai/deepseek-v4.1-flash", [Providers.Snowflake]: "snowflake/mistral-7b", [Providers.Tencent]: "tencent/deepseek-v4-pro", [Providers.Vertex_AI]: "gemini-pro",