diff --git a/.github/workflows/sync-aiand-models.yml b/.github/workflows/sync-aiand-models.yml new file mode 100644 index 00000000000..b476844b8d1 --- /dev/null +++ b/.github/workflows/sync-aiand-models.yml @@ -0,0 +1,73 @@ +name: Sync aiand model registry + +on: + schedule: + - cron: "45 7 * * *" + workflow_dispatch: + +concurrency: + group: sync-aiand-models + cancel-in-progress: false + +permissions: + contents: write + pull-requests: write + +jobs: + sync_aiand_models: + if: github.repository == 'BerriAI/litellm' + runs-on: ubuntu-latest + env: + BASE_BRANCH: ${{ github.event.repository.default_branch }} + steps: + - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + ref: ${{ env.BASE_BRANCH }} + persist-credentials: false + - name: Set up uv + uses: ./.github/actions/setup-uv-with-retries + with: + version: "0.10.9" + - name: Look for an already-open sync PR + id: existing + run: | + open_pr="$(gh pr list --repo "$GITHUB_REPOSITORY" --state open --limit 1000 --json headRefName,changedFiles \ + --jq '[.[] | select(.headRefName | startswith("litellm_aiand_registry_sync_")) | select(.changedFiles > 0)] | first | .headRefName // empty')" + echo "open_pr=$open_pr" >> "$GITHUB_OUTPUT" + if [ -n "$open_pr" ]; then + echo "Sync PR $open_pr still has unreviewed changes; skipping this run." + fi + env: + GH_TOKEN: ${{ secrets.GH_TOKEN || github.token }} + - name: Run the sync + if: steps.existing.outputs.open_pr == '' + run: | + uv run --frozen python scripts/sync_aiand_models.py --write --pr-body-file "$RUNNER_TEMP/pr_body.md" + - name: Regenerate the JSON schema + if: steps.existing.outputs.open_pr == '' + run: | + uv run --frozen python ci_cd/generate_model_prices_schema.py + - name: Create a pull request when the registry changed + if: steps.existing.outputs.open_pr == '' + run: | + if git diff --quiet; then + echo "Registry already in sync; no PR needed." + exit 0 + fi + branch="litellm_aiand_registry_sync_$(date +'%Y-%m-%d')" + git config user.name "github-actions[bot]" + git config user.email "41898282+github-actions[bot]@users.noreply.github.com" + git checkout -b "$branch" + git add model_prices_and_context_window.json \ + litellm/model_prices_and_context_window_backup.json \ + model_prices_and_context_window.schema.json + git commit -m "feat(models): sync aiand model registry $(date +'%Y-%m-%d')" + gh auth setup-git + git push origin --delete "$branch" 2>/dev/null || true + git push origin "$branch" + gh pr create --title "feat(models): sync aiand model registry" \ + --body-file "$RUNNER_TEMP/pr_body.md" \ + --head "$branch" \ + --base "$BASE_BRANCH" + env: + GH_TOKEN: ${{ secrets.GH_TOKEN || github.token }} diff --git a/README.md b/README.md index 7ffc44854bb..de13610141b 100644 --- a/README.md +++ b/README.md @@ -298,6 +298,7 @@ Set `LITELLM_PROXY_API_BASE` and `LITELLM_PROXY_API_KEY` and every model call th | Provider | `/chat/completions` | `/messages` | `/responses` | `/embeddings` | `/image/generations` | `/audio/transcriptions` | `/audio/speech` | `/moderations` | `/batches` | `/rerank` | |-------------------------------------------------------------------------------------|---------------------|-------------|--------------|---------------|----------------------|-------------------------|-----------------|----------------|-----------|-----------| | [Abliteration (`abliteration`)](https://docs.litellm.ai/docs/providers/abliteration) | ✅ | | | | | | | | | | +| [ai& (`aiand`)](https://docs.aiand.com/) | ✅ | ✅ | ✅ | | | | | | | | | [AI/ML API (`aiml`)](https://docs.litellm.ai/docs/providers/aiml) | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | | | [AI21 (`ai21`)](https://docs.litellm.ai/docs/providers/ai21) | ✅ | ✅ | ✅ | | | | | | | | | [AI21 Chat (`ai21_chat`)](https://docs.litellm.ai/docs/providers/ai21) | ✅ | ✅ | ✅ | | | | | | | | diff --git a/litellm/__init__.py b/litellm/__init__.py index fea7a27a5fb..6e800373e4d 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -686,6 +686,7 @@ qwencloud_models: Set = set() qwen_ai_platform_models: Set = set() moonshot_models: Set = set() publicai_models: Set = set() +aiand_models: Set[str] = set() darkbloom_models: Set = set() v0_models: Set = set() morph_models: Set = set() @@ -950,6 +951,8 @@ def _populate_provider_model_sets(model_cost_map: Dict) -> None: moonshot_models.add(key) elif value.get("litellm_provider") == "publicai": publicai_models.add(key) + elif value.get("litellm_provider") == "aiand": + aiand_models.add(key) elif value.get("litellm_provider") == "darkbloom": darkbloom_models.add(key) elif value.get("litellm_provider") == "v0": @@ -1114,6 +1117,7 @@ model_list = list( | qwen_ai_platform_models | moonshot_models | publicai_models + | aiand_models | darkbloom_models | v0_models | morph_models @@ -1226,6 +1230,7 @@ def _build_models_by_provider() -> dict: "modelscope": modelscope_models, "moonshot": moonshot_models, "publicai": publicai_models, + "aiand": aiand_models, "darkbloom": darkbloom_models, "v0": v0_models, "morph": morph_models, diff --git a/litellm/constants.py b/litellm/constants.py index d58fc8a6318..f89f54ad495 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -940,6 +940,7 @@ openai_compatible_endpoints: Final[list] = [ "https://dashscope.aliyuncs.com/compatible-mode/v1", "https://api-inference.modelscope.cn/v1", "https://api.moonshot.ai/v1", + "https://api.aiand.com/v1", "https://api.publicai.co/v1", "https://api.synthetic.new/openai/v1", "https://serverless.tensormesh.ai/v1", @@ -1002,6 +1003,7 @@ openai_compatible_providers: Final[list] = [ "chatgpt", # ChatGPT subscription API "novita", "meta_llama", + "aiand", "publicai", # PublicAI - JSON-configured provider "synthetic", # Synthetic - JSON-configured provider "tensormesh", # Tensormesh - JSON-configured provider @@ -1069,6 +1071,7 @@ openai_text_completion_compatible_providers: Final[list] = [ # providers that s "lambda_ai", "hyperbolic", "wandb", + "aiand", ] _openai_like_providers: Final[list] = [ "predibase", diff --git a/litellm/litellm_core_utils/exception_mapping_utils.py b/litellm/litellm_core_utils/exception_mapping_utils.py index 0fdfb301291..db8d4b81d40 100644 --- a/litellm/litellm_core_utils/exception_mapping_utils.py +++ b/litellm/litellm_core_utils/exception_mapping_utils.py @@ -1803,6 +1803,71 @@ def _map_together_ai_exception( ) +_AIAND_ERROR_PAYLOAD_KEYS: Final = ("message", "code", "type", "param") + + +def _map_aiand_exception( + *, + model: str, + original_exception: _ProviderHTTPException, + custom_llm_provider: str, + error_str: str, + exception_type: str, + exception_provider: str, + extra_information: str, +) -> None: + error_body: object = getattr(original_exception, "body", None) + if not isinstance(error_body, Mapping): + try: + error_body = json.loads(error_str) + except ValueError: + error_body = None + inner: Final[object] = error_body.get("error") if isinstance(error_body, Mapping) else None + error_payload: Final[Mapping[str, object]] = ( + inner + if isinstance(inner, Mapping) + else error_body + if isinstance(error_body, Mapping) and any(key in error_body for key in _AIAND_ERROR_PAYLOAD_KEYS) + else {} + ) + status_code: Final[int | None] = getattr(original_exception, "status_code", None) + error_message: Final[object] = error_payload.get("message") + message: Final[str] = error_message if isinstance(error_message, str) else error_str + if status_code == 402 and ( + error_payload.get("code") == "insufficient_credits" + or error_payload.get("type") == "billing_error" + or "insufficient_credits" in error_str + or "billing_error" in error_str + ): + raise PermissionDeniedError( + message=f"{exception_provider} - {message}", + llm_provider="aiand", + model=model, + response=_response_or_stub(original_exception, status_code=403), + litellm_debug_info=extra_information, + ) + elif status_code == 401 and (error_payload.get("code") == "invalid_api_key" or "invalid_api_key" in error_str): + raise AuthenticationError( + message=f"{exception_provider} - {message}", + llm_provider="aiand", + model=model, + response=getattr(original_exception, "response", None), + litellm_debug_info=extra_information, + ) + elif status_code == 404 and ( + error_payload.get("code") == "model_not_found" + or error_payload.get("param") == "model" + or "model_not_found" in error_str + ): + raise NotFoundError( + message=f"{exception_provider} - {message}", + model=model, + llm_provider="aiand", + response=getattr(original_exception, "response", None), + litellm_debug_info=extra_information, + ) + + def _map_aleph_alpha_exception( *, model: str, @@ -2463,6 +2528,17 @@ def exception_type( custom_llm_provider=custom_llm_provider, body=getattr(original_exception, "body", None), ) + if custom_llm_provider == "aiand": + _aiand_model: Final = model if isinstance(model, str) else "" + _map_aiand_exception( + model=_aiand_model, + original_exception=mappable_exception, + custom_llm_provider=custom_llm_provider, + error_str=error_str, + exception_type=exception_type, + exception_provider="AiandException", + extra_information=extra_information, + ) if ( custom_llm_provider == "openai" or custom_llm_provider == "text-completion-openai" diff --git a/litellm/llms/openai_like/providers.json b/litellm/llms/openai_like/providers.json index 61ff4be3a46..847ab731c68 100644 --- a/litellm/llms/openai_like/providers.json +++ b/litellm/llms/openai_like/providers.json @@ -107,6 +107,12 @@ "api_key_env": "AIHUBMIX_API_KEY", "api_base_env": "AIHUBMIX_API_BASE" }, + "aiand": { + "base_url": "https://api.aiand.com/v1", + "api_key_env": "AIAND_API_KEY", + "api_base_env": "AIAND_API_BASE", + "supported_endpoints": ["/v1/chat/completions", "/v1/responses", "/v1/messages", "/v1/completions"] + }, "crusoe": { "base_url": "https://managed-inference-api-proxy.crusoecloud.com/v1", "api_key_env": "CRUSOE_API_KEY", diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index cd38b5b0d21..27005e8174f 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -59714,6 +59714,392 @@ ], "supports_audio_input": true }, + "aiand/deepseek-ai/deepseek-v4-flash": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 2.5e-07, + "cache_read_input_token_cost": 8e-08, + "max_input_tokens": 1048576, + "max_output_tokens": 384000, + "max_tokens": 384000, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ], + "reasoning_effort_levels": [ + "none", + "high", + "max" + ] + }, + "aiand/deepseek-ai/deepseek-v4-pro": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1e-06, + "output_cost_per_token": 2.5e-06, + "cache_read_input_token_cost": 2.5e-07, + "max_input_tokens": 1048576, + "max_output_tokens": 384000, + "max_tokens": 384000, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ], + "reasoning_effort_levels": [ + "none", + "high", + "max" + ] + }, + "aiand/deepseek-ai/deepseek-v4.1-flash": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 3e-07, + "output_cost_per_token": 6e-07, + "cache_read_input_token_cost": 2e-08, + "max_input_tokens": 1048576, + "max_output_tokens": 384000, + "max_tokens": 384000, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ], + "reasoning_effort_levels": [ + "none", + "high", + "max" + ] + }, + "aiand/google/gemma-4-31b-it": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 2e-07, + "output_cost_per_token": 5e-07, + "cache_read_input_token_cost": 5e-08, + "max_input_tokens": 262144, + "max_output_tokens": 32768, + "max_tokens": 32768, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ], + "reasoning_effort_levels": [ + "none", + "high" + ] + }, + "aiand/moonshotai/kimi-k2.7-code": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 7.5e-07, + "output_cost_per_token": 3.5e-06, + "cache_read_input_token_cost": 2e-07, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "max_tokens": 262144, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ], + "reasoning_effort_levels": [ + "high" + ] + }, + "aiand/moonshotai/kimi-k3": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 3e-06, + "output_cost_per_token": 1.25e-05, + "cache_read_input_token_cost": 5e-07, + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "max_tokens": 131072, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ], + "reasoning_effort_levels": [ + "low", + "high", + "max" + ] + }, + "aiand/motif-technologies/motif-3": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 5e-07, + "output_cost_per_token": 2e-06, + "cache_read_input_token_cost": 2e-07, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "max_tokens": 262144, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": false, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ], + "reasoning_effort_levels": [ + "none", + "high" + ] + }, + "aiand/openai/gpt-oss-120b": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 6e-07, + "cache_read_input_token_cost": 8e-08, + "max_input_tokens": 131072, + "max_output_tokens": 32768, + "max_tokens": 32768, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ], + "reasoning_effort_levels": [ + "low", + "medium", + "high" + ] + }, + "aiand/qwen/qwen3.6-27b": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 3.2e-07, + "output_cost_per_token": 3.2e-06, + "cache_read_input_token_cost": 2e-07, + "max_input_tokens": 262144, + "max_output_tokens": 65536, + "max_tokens": 65536, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ], + "reasoning_effort_levels": [ + "none", + "high" + ] + }, + "aiand/qwen/qwen3.8-27b": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 4e-07, + "output_cost_per_token": 3e-06, + "cache_read_input_token_cost": 2e-07, + "max_input_tokens": 262144, + "max_output_tokens": 32768, + "max_tokens": 32768, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ], + "reasoning_effort_levels": [ + "none", + "low", + "medium", + "xhigh" + ] + }, + "aiand/zai-org/glm-5.2": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1e-06, + "output_cost_per_token": 4e-06, + "cache_read_input_token_cost": 3e-07, + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "max_tokens": 131072, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ], + "reasoning_effort_levels": [ + "none", + "high", + "max" + ] + }, + "aiand/zai-org/glm-5.3": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1e-06, + "output_cost_per_token": 4e-06, + "cache_read_input_token_cost": 3e-07, + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "max_tokens": 131072, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ], + "reasoning_effort_levels": [ + "low", + "high", + "max" + ] + }, + "aiand/zai-org/glm-5.3-flash": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 5e-07, + "cache_read_input_token_cost": 3e-08, + "max_input_tokens": 1048550, + "max_output_tokens": 131072, + "max_tokens": 131072, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ], + "reasoning_effort_levels": [ + "low", + "high", + "max" + ] + }, "tensormesh/Qwen/Qwen3.5-397B-A17B-FP8": { "litellm_provider": "tensormesh", "mode": "chat", diff --git a/litellm/provider_endpoints_support_backup.json b/litellm/provider_endpoints_support_backup.json index c9635587eeb..2694af5ada1 100644 --- a/litellm/provider_endpoints_support_backup.json +++ b/litellm/provider_endpoints_support_backup.json @@ -120,6 +120,24 @@ "interactions": true } }, + "aiand": { + "display_name": "ai& (`aiand`)", + "url": "https://docs.aiand.com/api/chat-completions/", + "endpoints": { + "chat_completions": true, + "messages": true, + "responses": true, + "embeddings": false, + "image_generations": false, + "audio_transcriptions": false, + "audio_speech": false, + "moderations": false, + "batches": false, + "rerank": false, + "a2a": false, + "interactions": false + } + }, "amazon_nova": { "display_name": "Amazon Nova (`amazon_nova`)", "url": "https://docs.litellm.ai/docs/providers/amazon_nova", diff --git a/litellm/proxy/public_endpoints/provider_create_fields.json b/litellm/proxy/public_endpoints/provider_create_fields.json index 6e96d6ad0ec..2a8995d0db1 100644 --- a/litellm/proxy/public_endpoints/provider_create_fields.json +++ b/litellm/proxy/public_endpoints/provider_create_fields.json @@ -1,4 +1,32 @@ [ + { + "provider": "AIAND", + "provider_display_name": "ai&", + "litellm_provider": "aiand", + "credential_fields": [ + { + "key": "api_base", + "label": "API Base", + "placeholder": "https://api.aiand.com/v1", + "tooltip": null, + "required": false, + "field_type": "text", + "options": null, + "default_value": null + }, + { + "key": "api_key", + "label": "API Key", + "placeholder": null, + "tooltip": null, + "required": true, + "field_type": "password", + "options": null, + "default_value": null + } + ], + "default_model_placeholder": "aiand/deepseek-ai/deepseek-v4.1-flash" + }, { "provider": "AIML", "provider_display_name": "AI/ML API", diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 142f9b14a72..44871fecbe8 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -4171,6 +4171,7 @@ class LlmProviders(str, Enum): PARASAIL = "parasail" XIAOMI_MIMO = "xiaomi_mimo" TENSORMESH = "tensormesh" + AIAND = "aiand" LIBERTAI = "libertai" PINSTRIPES = "pinstripes" COGNITION = "cognition" diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index cd38b5b0d21..27005e8174f 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -59714,6 +59714,392 @@ ], "supports_audio_input": true }, + "aiand/deepseek-ai/deepseek-v4-flash": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 2.5e-07, + "cache_read_input_token_cost": 8e-08, + "max_input_tokens": 1048576, + "max_output_tokens": 384000, + "max_tokens": 384000, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ], + "reasoning_effort_levels": [ + "none", + "high", + "max" + ] + }, + "aiand/deepseek-ai/deepseek-v4-pro": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1e-06, + "output_cost_per_token": 2.5e-06, + "cache_read_input_token_cost": 2.5e-07, + "max_input_tokens": 1048576, + "max_output_tokens": 384000, + "max_tokens": 384000, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ], + "reasoning_effort_levels": [ + "none", + "high", + "max" + ] + }, + "aiand/deepseek-ai/deepseek-v4.1-flash": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 3e-07, + "output_cost_per_token": 6e-07, + "cache_read_input_token_cost": 2e-08, + "max_input_tokens": 1048576, + "max_output_tokens": 384000, + "max_tokens": 384000, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ], + "reasoning_effort_levels": [ + "none", + "high", + "max" + ] + }, + "aiand/google/gemma-4-31b-it": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 2e-07, + "output_cost_per_token": 5e-07, + "cache_read_input_token_cost": 5e-08, + "max_input_tokens": 262144, + "max_output_tokens": 32768, + "max_tokens": 32768, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ], + "reasoning_effort_levels": [ + "none", + "high" + ] + }, + "aiand/moonshotai/kimi-k2.7-code": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 7.5e-07, + "output_cost_per_token": 3.5e-06, + "cache_read_input_token_cost": 2e-07, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "max_tokens": 262144, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ], + "reasoning_effort_levels": [ + "high" + ] + }, + "aiand/moonshotai/kimi-k3": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 3e-06, + "output_cost_per_token": 1.25e-05, + "cache_read_input_token_cost": 5e-07, + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "max_tokens": 131072, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ], + "reasoning_effort_levels": [ + "low", + "high", + "max" + ] + }, + "aiand/motif-technologies/motif-3": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 5e-07, + "output_cost_per_token": 2e-06, + "cache_read_input_token_cost": 2e-07, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "max_tokens": 262144, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": false, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ], + "reasoning_effort_levels": [ + "none", + "high" + ] + }, + "aiand/openai/gpt-oss-120b": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 6e-07, + "cache_read_input_token_cost": 8e-08, + "max_input_tokens": 131072, + "max_output_tokens": 32768, + "max_tokens": 32768, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ], + "reasoning_effort_levels": [ + "low", + "medium", + "high" + ] + }, + "aiand/qwen/qwen3.6-27b": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 3.2e-07, + "output_cost_per_token": 3.2e-06, + "cache_read_input_token_cost": 2e-07, + "max_input_tokens": 262144, + "max_output_tokens": 65536, + "max_tokens": 65536, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ], + "reasoning_effort_levels": [ + "none", + "high" + ] + }, + "aiand/qwen/qwen3.8-27b": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 4e-07, + "output_cost_per_token": 3e-06, + "cache_read_input_token_cost": 2e-07, + "max_input_tokens": 262144, + "max_output_tokens": 32768, + "max_tokens": 32768, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ], + "reasoning_effort_levels": [ + "none", + "low", + "medium", + "xhigh" + ] + }, + "aiand/zai-org/glm-5.2": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1e-06, + "output_cost_per_token": 4e-06, + "cache_read_input_token_cost": 3e-07, + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "max_tokens": 131072, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ], + "reasoning_effort_levels": [ + "none", + "high", + "max" + ] + }, + "aiand/zai-org/glm-5.3": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1e-06, + "output_cost_per_token": 4e-06, + "cache_read_input_token_cost": 3e-07, + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "max_tokens": 131072, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": false, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ], + "reasoning_effort_levels": [ + "low", + "high", + "max" + ] + }, + "aiand/zai-org/glm-5.3-flash": { + "litellm_provider": "aiand", + "mode": "chat", + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 5e-07, + "cache_read_input_token_cost": 3e-08, + "max_input_tokens": 1048550, + "max_output_tokens": 131072, + "max_tokens": 131072, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://api.aiand.com/v1/api.json", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses", + "/v1/messages" + ], + "reasoning_effort_levels": [ + "low", + "high", + "max" + ] + }, "tensormesh/Qwen/Qwen3.5-397B-A17B-FP8": { "litellm_provider": "tensormesh", "mode": "chat", diff --git a/provider_endpoints_support.json b/provider_endpoints_support.json index 7ffaacdb3aa..ad5c92d9a1a 100644 --- a/provider_endpoints_support.json +++ b/provider_endpoints_support.json @@ -121,6 +121,24 @@ "interactions": true } }, + "aiand": { + "display_name": "ai& (`aiand`)", + "url": "https://docs.aiand.com/api/chat-completions/", + "endpoints": { + "chat_completions": true, + "messages": true, + "responses": true, + "embeddings": false, + "image_generations": false, + "audio_transcriptions": false, + "audio_speech": false, + "moderations": false, + "batches": false, + "rerank": false, + "a2a": false, + "interactions": false + } + }, "amazon_nova": { "display_name": "Amazon Nova (`amazon_nova`)", "url": "https://docs.litellm.ai/docs/providers/amazon_nova", diff --git a/scripts/sync_aiand_models.py b/scripts/sync_aiand_models.py new file mode 100644 index 00000000000..06e589eeaa5 --- /dev/null +++ b/scripts/sync_aiand_models.py @@ -0,0 +1,327 @@ +"""Sync the aiand entries of model_prices_and_context_window.json with aiand's live model spec. + +Pulls ``GET https://api.aiand.com/v1/api.json`` (public, no auth), maps spec fields onto +registry fields, and diffs the result against the registry. Dry run (the default) prints the +diff summary and the generated PR body; ``--write`` applies the changes to the root cost map +and its ``litellm/`` backup copy. + +Policy highlights: +- Prices arrive per 1M tokens with float artifacts and are normalized to clean per-token values. +- Registry entries are never deleted; a model absent from the live spec is stamped with + ``metadata.absent_from_spec_since`` (a real, PR-worthy file change) and surfaced as a + warning for a human deprecation call; the stamp is cleared when the model reappears in + the spec. +- The spec cannot express endpoint support or caching behavior, so ``supported_endpoints`` and + ``supports_prompt_caching`` stay fixed for the whole provider. +- Reasoning effort levels map from the spec's ``effort`` reasoning option, when one is declared. +""" + +import argparse +import json +import sys +from collections.abc import Mapping, Sequence +from dataclasses import dataclass +from datetime import datetime, timezone +from pathlib import Path +from typing import Final + +import httpx +from pydantic import BaseModel, TypeAdapter, ValidationError + +SPEC_URL: Final = "https://api.aiand.com/v1/api.json" +PROVIDER: Final = "aiand" +PREFIX: Final = "aiand/" +SOURCE_URL: Final = "https://api.aiand.com/v1/api.json" +SUPPORTED_ENDPOINTS: Final = ("/v1/chat/completions", "/v1/responses", "/v1/messages") +COST_MAP_RELPATHS: Final = ( + "model_prices_and_context_window.json", + "litellm/model_prices_and_context_window_backup.json", +) + + +class SyncError(RuntimeError): + pass + + +class SpecCost(BaseModel): + input: float + output: float + cache_read: float + + +class SpecLimit(BaseModel): + context: int + output: int + + +class SpecModalities(BaseModel): + input: list[str] + + +class SpecReasoningOption(BaseModel): + type: str + values: list[str] + + +class SpecModel(BaseModel): + id: str + name: str + family: str + reasoning: bool + reasoning_options: list[SpecReasoningOption] = [] + tool_call: bool + structured_output: bool + temperature: bool + attachment: bool + open_weights: bool + cost: SpecCost + limit: SpecLimit + modalities: SpecModalities + + +SPEC_ADAPTER: Final = TypeAdapter(dict[str, SpecModel]) + +RegistryEntry = dict[str, object] +CostMap = dict[str, object] + + +@dataclass(frozen=True, slots=True) +class SyncOutcome: + cost_map: CostMap + added: tuple[str, ...] = () + updated: tuple[str, ...] = () + removed: tuple[str, ...] = () + warnings: tuple[str, ...] = () + + @property + def has_changes(self) -> bool: + return bool(self.added or self.updated or self.removed or self.warnings) + + +def per_token(price_per_million: float) -> float: + return float(f"{price_per_million / 1e6:.6g}") + + +def _today() -> str: + return datetime.now(tz=timezone.utc).date().isoformat() + + +def _effort_levels(model: SpecModel) -> tuple[str, ...]: + for option in model.reasoning_options: + if option.type == "effort": + return tuple(option.values) + return () + + +def _spec_fields(model: SpecModel) -> RegistryEntry: + effort_levels: Final = _effort_levels(model) + fields: RegistryEntry = { + "litellm_provider": PROVIDER, + "mode": "chat", + "input_cost_per_token": per_token(model.cost.input), + "output_cost_per_token": per_token(model.cost.output), + "cache_read_input_token_cost": per_token(model.cost.cache_read), + "max_input_tokens": model.limit.context, + "max_output_tokens": model.limit.output, + "max_tokens": model.limit.output, + "supports_function_calling": model.tool_call, + "supports_native_streaming": True, + "supports_parallel_function_calling": model.tool_call, + "supports_tool_choice": model.tool_call, + "supports_response_schema": model.structured_output, + "supports_prompt_caching": True, + "supports_system_messages": True, + "supports_reasoning": model.reasoning, + "supports_vision": "image" in model.modalities.input, + "source": SOURCE_URL, + "supported_endpoints": list(SUPPORTED_ENDPOINTS), + } + fields["reasoning_effort_levels"] = list(effort_levels) + return fields + + +def _new_entry(model: SpecModel) -> RegistryEntry: + return _spec_fields(model) + + +def _updated_entry(entry: RegistryEntry, model: SpecModel) -> tuple[RegistryEntry, tuple[str, ...]]: + desired: Final = _spec_fields(model) + changes: Final = tuple( + f"{name}: {entry.get(name)!r} -> {value!r}" for name, value in desired.items() if entry.get(name) != value + ) + extras: Final = dict(sorted((name, value) for name, value in entry.items() if name not in desired)) + return {**desired, **extras}, changes + + +def _with_new_keys_in_block(original: CostMap, result: CostMap, new_keys: Sequence[str]) -> CostMap: + provider_keys: Final = tuple(key for key in original if key.startswith(PREFIX)) + if not new_keys or not provider_keys: + return result + block_end: Final = provider_keys[-1] + return { + key: value + for existing in original + for key, value in ( + (existing, result[existing]), + *((new, result[new]) for new in sorted(new_keys) if existing == block_end), + ) + } + + +def compute_sync(cost_map: CostMap, spec: Mapping[str, SpecModel]) -> SyncOutcome: + spec_ids: Final = frozenset(spec) + registry_ids: Final = {key.removeprefix(PREFIX): key for key in cost_map if key.startswith(PREFIX)} + + added: Final[list[str]] = [] + updated: Final[list[str]] = [] + removed: Final[list[str]] = [] + warnings: Final[list[str]] = [] + result: Final[CostMap] = dict(cost_map) + + for model_id, model in sorted(spec.items()): + key: Final = f"{PREFIX}{model_id}" + entry = result.get(key) + if not isinstance(entry, dict): + result[key] = _new_entry(model) + added.append(key) + continue + new_entry, changes = _updated_entry(entry, model) + metadata = new_entry.get("metadata") + stamped = metadata.get("absent_from_spec_since") if isinstance(metadata, dict) else None + if stamped is not None: + remaining_metadata: Final = { + name: value for name, value in metadata.items() if name != "absent_from_spec_since" + } + if remaining_metadata: + new_entry["metadata"] = dict(sorted(remaining_metadata.items())) + else: + new_entry.pop("metadata") + changes = ( + *changes, + f"metadata.absent_from_spec_since: {stamped!r} -> None (model reappeared in the spec)", + ) + if changes: + updated.append(f"{key}: " + "; ".join(changes)) + result[key] = new_entry + + for model_id, key in sorted(registry_ids.items()): + if model_id in spec_ids: + continue + removed.append(key) + warnings.append( + f"`{key}` is absent from the live spec; the registry entry is kept (never deleted) " + "and needs a human deprecation call" + ) + entry = result.get(key) + if not isinstance(entry, dict): + continue + metadata = entry.get("metadata") + curated: Final = metadata.get("absent_from_spec_since") if isinstance(metadata, dict) else None + if curated is not None: + continue + extras: Final = dict(metadata) if isinstance(metadata, dict) else {} + stamped_entry: Final = dict(entry) + stamped_entry["metadata"] = dict(sorted({**extras, "absent_from_spec_since": _today()}.items())) + result[key] = dict(sorted(stamped_entry.items())) + updated.append(f"{key}: metadata.absent_from_spec_since: None -> {_today()!r}") + return SyncOutcome( + cost_map=_with_new_keys_in_block(cost_map, result, tuple(added)), + added=tuple(added), + updated=tuple(updated), + removed=tuple(removed), + warnings=tuple(warnings), + ) + + +def _section_block(title: str, lines: Sequence[str], backtick: bool) -> str: + bullets: Final = "\n".join(f"- `{line}`" if backtick else f"- {line}" for line in lines) or "- none" + return f"### {title} ({len(lines)})\n{bullets}\n" + + +def render_pr_body(outcome: SyncOutcome) -> str: + return ( + "Automated daily sync of the aiand entries in model_prices_and_context_window.json against " + f"`GET {SPEC_URL}` by scripts/sync_aiand_models.py.\n" + "\n" + f"{_section_block('Added', outcome.added, backtick=True)}" + "\n" + f"{_section_block('Updated', outcome.updated, backtick=True)}" + "\n" + f"{_section_block('Removed from the spec', outcome.warnings, backtick=False)}" + ) + + +def render_summary(outcome: SyncOutcome) -> str: + return ( + f"added={len(outcome.added)} updated={len(outcome.updated)} " + f"removed={len(outcome.removed)} warnings={len(outcome.warnings)}" + ) + + +def load_spec(raw: bytes) -> dict[str, SpecModel]: + parsed: Final = json.loads(raw) + provider: Final = parsed.get("aiand") if isinstance(parsed, dict) else None + models: Final = provider.get("models") if isinstance(provider, dict) else None + try: + spec: Final = SPEC_ADAPTER.validate_python(models) + except ValidationError as error: + raise SyncError(f"the spec response no longer matches the expected shape: {error}") from error + if not spec: + raise SyncError("the spec response contains no aiand models; refusing to rewrite the registry") + for model_id, model in spec.items(): + if model.id != model_id: + raise SyncError(f"spec model id {model.id!r} does not match its key {model_id!r}") + return spec + + +def _fetch(url: str) -> bytes: + response: Final = httpx.get(url, timeout=30, follow_redirects=True) + if response.status_code != 200: + raise SyncError(f"GET {url} returned {response.status_code}") + return response.content + + +def _serialize(cost_map: CostMap) -> str: + return json.dumps(cost_map, indent=4, ensure_ascii=False) + "\n" + + +def main(argv: Sequence[str]) -> int: + parser: Final = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--write", action="store_true", help="apply the sync to the cost map files (default: dry run)") + parser.add_argument("--spec-json", type=Path, help="recorded spec response to use instead of the live API") + parser.add_argument("--pr-body-file", type=Path, help="write the generated PR body to this path") + parser.add_argument("--repo-root", type=Path, default=Path(__file__).resolve().parent.parent) + args: Final = parser.parse_args(argv) + + if args.spec_json is not None: + spec_raw: Final = args.spec_json.read_bytes() + else: + spec_raw = _fetch(SPEC_URL) # rebind-ok: branch-dependent source + spec: Final = load_spec(spec_raw) + + cost_map_path: Final = args.repo_root / COST_MAP_RELPATHS[0] + cost_map: Final = json.loads(cost_map_path.read_text()) + outcome: Final = compute_sync(cost_map, spec) + body: Final = render_pr_body(outcome) + + if args.pr_body_file is not None and outcome.has_changes: + args.pr_body_file.write_text(body) + if args.write and (outcome.added or outcome.updated): + for relpath in COST_MAP_RELPATHS: + (args.repo_root / relpath).write_text(_serialize(outcome.cost_map)) + print(render_summary(outcome)) + print() + print(body) + if not args.write: + print("dry run: no files were touched") + elif not (outcome.added or outcome.updated): + print("registry already in sync: no files were touched") + return 0 + + +if __name__ == "__main__": + try: + raise SystemExit(main(sys.argv[1:])) + except SyncError as error: + print(f"SYNC FAILED: {error}", file=sys.stderr) + raise SystemExit(1) from error diff --git a/tests/unit/litellm_core_utils/test_exception_mapping_utils.py b/tests/unit/litellm_core_utils/test_exception_mapping_utils.py index 9de768ea47b..3f936e3e77c 100644 --- a/tests/unit/litellm_core_utils/test_exception_mapping_utils.py +++ b/tests/unit/litellm_core_utils/test_exception_mapping_utils.py @@ -1,3 +1,5 @@ +import json + import httpx import openai import pytest @@ -9,6 +11,7 @@ from litellm.litellm_core_utils.exception_mapping_utils import ( ExceptionCheckers, _get_body_error_code, _get_response_headers, + _map_aiand_exception, exception_type, extract_and_raise_litellm_exception, ) @@ -1039,6 +1042,262 @@ def test_an_unmapped_exception_with_no_model_or_provider_message_keeps_traceback assert "Traceback (most recent call last)" in raised.value.message +AIAND_INSUFFICIENT_CREDITS_MESSAGE = ( + "Insufficient credits. Review billing at https://console.aiand.com/settings/billing to continue." +) + + +@pytest.mark.parametrize( + "error_body", + [ + { + "message": AIAND_INSUFFICIENT_CREDITS_MESSAGE, + "type": "billing_error", + "param": None, + "code": "insufficient_credits", + }, + { + "message": "Balance too low to process the request.", + "type": "billing_error", + "param": None, + "code": None, + }, + ], +) +def test_an_aiand_402_billing_error_is_a_permission_denied_error(error_body, quiet_exception_mapping): + from litellm.llms.base_llm.chat.transformation import BaseLLMException + + original_exception = BaseLLMException( + status_code=402, + message=json.dumps({"error": error_body}), + ) + + with pytest.raises(litellm.PermissionDeniedError) as raised: + exception_type( + model="test-model", + original_exception=original_exception, + custom_llm_provider="aiand", + ) + + assert raised.value.status_code == 402 + assert raised.value.llm_provider == "aiand" + assert raised.value.model == "test-model" + assert raised.value.message == f"litellm.PermissionDeniedError: AiandException - {error_body['message']}" + + +def test_an_aiand_402_from_the_response_body_is_a_permission_denied_error(quiet_exception_mapping): + from litellm.llms.base_llm.chat.transformation import BaseLLMException + + original_exception = BaseLLMException( + status_code=402, + message=AIAND_INSUFFICIENT_CREDITS_MESSAGE, + body={ + "error": { + "message": AIAND_INSUFFICIENT_CREDITS_MESSAGE, + "type": "billing_error", + "param": None, + "code": "insufficient_credits", + } + }, + ) + + with pytest.raises(litellm.PermissionDeniedError) as raised: + exception_type( + model="test-model", + original_exception=original_exception, + custom_llm_provider="aiand", + ) + + assert raised.value.llm_provider == "aiand" + assert ( + raised.value.message == f"litellm.PermissionDeniedError: AiandException - {AIAND_INSUFFICIENT_CREDITS_MESSAGE}" + ) + + +def test_an_aiand_402_from_an_unwrapped_body_is_a_permission_denied_error(quiet_exception_mapping): + from litellm.llms.base_llm.chat.transformation import BaseLLMException + + original_exception = BaseLLMException( + status_code=402, + message=( + "Error code: 402 - {'error': {'message': " + f"'{AIAND_INSUFFICIENT_CREDITS_MESSAGE}', " + "'type': 'billing_error', 'param': None, 'code': 'insufficient_credits'}}" + ), + body={ + "message": AIAND_INSUFFICIENT_CREDITS_MESSAGE, + "type": "billing_error", + "param": None, + "code": "insufficient_credits", + }, + ) + + with pytest.raises(litellm.PermissionDeniedError) as raised: + exception_type( + model="test-model", + original_exception=original_exception, + custom_llm_provider="aiand", + ) + + assert raised.value.status_code == 402 + assert raised.value.llm_provider == "aiand" + assert ( + raised.value.message == f"litellm.PermissionDeniedError: AiandException - {AIAND_INSUFFICIENT_CREDITS_MESSAGE}" + ) + + +def test_an_aiand_402_from_a_non_json_error_str_is_a_permission_denied_error(quiet_exception_mapping): + from litellm.llms.base_llm.chat.transformation import BaseLLMException + + error_str = ( + "Error code: 402 - {'error': {'message': " + f"'{AIAND_INSUFFICIENT_CREDITS_MESSAGE}', " + "'type': 'billing_error', 'param': None, 'code': 'insufficient_credits'}}" + ) + original_exception = BaseLLMException(status_code=402, message=error_str) + + with pytest.raises(litellm.PermissionDeniedError) as raised: + exception_type( + model="test-model", + original_exception=original_exception, + custom_llm_provider="aiand", + ) + + assert raised.value.status_code == 402 + assert raised.value.llm_provider == "aiand" + assert raised.value.message == f"litellm.PermissionDeniedError: AiandException - {error_str}" + + +def test_an_aiand_401_invalid_api_key_is_an_authentication_error(quiet_exception_mapping): + from litellm.llms.base_llm.chat.transformation import BaseLLMException + + original_exception = BaseLLMException( + status_code=401, + message=json.dumps( + { + "error": { + "message": "Missing or invalid API key", + "type": "authentication_error", + "param": None, + "code": "invalid_api_key", + } + } + ), + ) + + with pytest.raises(litellm.AuthenticationError) as raised: + exception_type( + model="test-model", + original_exception=original_exception, + custom_llm_provider="aiand", + ) + + assert raised.value.status_code == 401 + assert raised.value.llm_provider == "aiand" + assert raised.value.model == "test-model" + assert raised.value.message == "litellm.AuthenticationError: AiandException - Missing or invalid API key" + + +def test_an_aiand_404_model_not_found_is_a_not_found_error(quiet_exception_mapping): + from litellm.llms.base_llm.chat.transformation import BaseLLMException + + original_exception = BaseLLMException( + status_code=404, + message=json.dumps( + { + "error": { + "message": "Model not found", + "type": "invalid_request_error", + "param": "model", + "code": "model_not_found", + } + } + ), + ) + + with pytest.raises(litellm.NotFoundError) as raised: + exception_type( + model="test-model", + original_exception=original_exception, + custom_llm_provider="aiand", + ) + + assert raised.value.status_code == 404 + assert raised.value.llm_provider == "aiand" + assert raised.value.model == "test-model" + assert raised.value.message == "litellm.NotFoundError: AiandException - Model not found" + + +def test_an_aiand_context_window_error_is_a_context_window_exceeded_error(quiet_exception_mapping): + from litellm.llms.base_llm.chat.transformation import BaseLLMException + + original_exception = BaseLLMException(status_code=400, message=CONTEXT_WINDOW_MESSAGE) + + with pytest.raises(litellm.ContextWindowExceededError) as raised: + exception_type( + model="test-model", + original_exception=original_exception, + custom_llm_provider="aiand", + ) + + assert type(raised.value) is litellm.ContextWindowExceededError + assert raised.value.status_code == 400 + assert raised.value.llm_provider == "aiand" + assert raised.value.model == "test-model" + assert f"ContextWindowExceededError: AiandException - {CONTEXT_WINDOW_MESSAGE}" in raised.value.message + + +def test_an_unknown_aiand_error_falls_through_without_raising(): + from litellm.llms.base_llm.chat.transformation import BaseLLMException + + class _AiandUpstreamError(BaseLLMException): + code = "" + llm_provider = "aiand" + + original_exception = _AiandUpstreamError(status_code=418, message="I am a teapot") + + assert ( + _map_aiand_exception( + model="test-model", + original_exception=original_exception, + custom_llm_provider="aiand", + error_str="I am a teapot", + exception_type="HTTPException", + exception_provider="AiandException", + extra_information="", + ) + is None + ) + + +def test_an_unknown_aiand_error_still_maps_by_the_upstream_status(quiet_exception_mapping): + from litellm.llms.base_llm.chat.transformation import BaseLLMException + + original_exception = BaseLLMException( + status_code=400, + message=json.dumps( + { + "error": { + "message": "Something else went wrong", + "type": "invalid_request_error", + "param": None, + "code": "invalid_value", + } + } + ), + ) + + with pytest.raises(litellm.BadRequestError) as raised: + exception_type( + model="test-model", + original_exception=original_exception, + custom_llm_provider="aiand", + ) + + assert raised.value.status_code == 400 + assert raised.value.llm_provider == "aiand" + + CONTEXT_WINDOW_MESSAGE = "This model's maximum context length is 4096 tokens." CONTENT_POLICY_MESSAGE = '{"error": {"type": "invalid_request_error", "code": "content_policy_violation"}}' TIMEOUT_MESSAGE = "Request timed out." diff --git a/tests/unit/llms/openai_like/test_aiand_provider.py b/tests/unit/llms/openai_like/test_aiand_provider.py new file mode 100644 index 00000000000..72637b119ad --- /dev/null +++ b/tests/unit/llms/openai_like/test_aiand_provider.py @@ -0,0 +1,330 @@ +""" +Tests for aiand provider configuration and integration. +""" + +import json +from pathlib import Path +from typing import Final, get_args + +import pytest +import respx + +import litellm +from litellm.caching.llm_caching_handler import LLMClientCache +from litellm.types.llms.openai import REASONING_EFFORT + + +def test_aiand_provider_resolution(monkeypatch: pytest.MonkeyPatch) -> None: + from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider + + monkeypatch.setenv("AIAND_API_KEY", "aiand-test-key") + + model, provider, api_key, api_base = get_llm_provider( + model="aiand/deepseek-ai/deepseek-v4.1-flash", + custom_llm_provider=None, + api_base=None, + api_key=None, + ) + + assert model == "deepseek-ai/deepseek-v4.1-flash" + assert provider == "aiand" + assert api_key == "aiand-test-key" + assert api_base == "https://api.aiand.com/v1" + + +def test_aiand_provider_keeps_explicit_credentials(monkeypatch: pytest.MonkeyPatch) -> None: + from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider + + monkeypatch.setenv("AIAND_API_KEY", "aiand-env-key") + + _, provider, api_key, api_base = get_llm_provider( + model="aiand/deepseek-ai/deepseek-v4.1-flash", + custom_llm_provider=None, + api_base="https://aiand.internal.example/v1", + api_key="aiand-explicit-key", + ) + + assert provider == "aiand" + assert api_key == "aiand-explicit-key" + assert api_base == "https://aiand.internal.example/v1" + + +def test_aiand_url_autodetection(monkeypatch: pytest.MonkeyPatch) -> None: + from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider + + monkeypatch.setenv("AIAND_API_KEY", "aiand-test-key") + + model, provider, api_key, api_base = get_llm_provider( + model="deepseek-v4.1-flash", + custom_llm_provider=None, + api_base="https://api.aiand.com/v1", + api_key=None, + ) + + assert model == "deepseek-v4.1-flash" + assert provider == "aiand" + assert api_key == "aiand-test-key" + assert api_base == "https://api.aiand.com/v1" + + +AIAND_MODELS = tuple(sorted(name for name in litellm.model_cost if name.startswith("aiand/"))) + + +@pytest.mark.parametrize("model", AIAND_MODELS) +def test_aiand_model_cost_and_capabilities(model: str) -> None: + from litellm.cost_calculator import cost_per_token + + prompt_cost, completion_cost = cost_per_token( + model=model, + prompt_tokens=1_000_000, + completion_tokens=1_000_000, + custom_llm_provider="aiand", + ) + model_info = litellm.get_model_info(model) + + assert prompt_cost == pytest.approx(model_info["input_cost_per_token"] * 1_000_000) + assert completion_cost == pytest.approx(model_info["output_cost_per_token"] * 1_000_000) + assert 0 < model_info["cache_read_input_token_cost"] < model_info["input_cost_per_token"] + assert model_info["output_cost_per_token"] > 0 + assert model_info["max_tokens"] == model_info["max_output_tokens"] <= model_info["max_input_tokens"] + assert model_info["litellm_provider"] == "aiand" + assert model_info["mode"] == "chat" + assert type(model_info["supports_function_calling"]) is bool + assert type(model_info["supports_native_streaming"]) is bool + assert type(model_info["supports_reasoning"]) is bool + assert type(model_info["supports_response_schema"]) is bool + assert litellm.supports_vision(model) is model_info["supports_vision"] + + +def test_aiand_reasoning_effort_levels_are_valid() -> None: + known_efforts: Final = frozenset(get_args(REASONING_EFFORT)) + for model in AIAND_MODELS: + model_info = litellm.get_model_info(model) + levels = model_info.get("reasoning_effort_levels", []) + assert set(levels) <= known_efforts, f"{model} declares unknown reasoning efforts" + if levels: + assert model_info["supports_reasoning"] is True, ( + f"{model} declares reasoning efforts without supports_reasoning" + ) + + +def test_aiand_backup_registry_mirrors_cost_map() -> None: + package_root = Path(litellm.__file__).parent + cost_map = json.loads((package_root.parent / "model_prices_and_context_window.json").read_text()) + backup = json.loads((package_root / "model_prices_and_context_window_backup.json").read_text()) + aiand_entries = {name: entry for name, entry in cost_map.items() if name.startswith("aiand/")} + + assert tuple(sorted(aiand_entries)) == AIAND_MODELS + assert aiand_entries + assert all("supports_vision" in entry for entry in aiand_entries.values()) + assert aiand_entries == {name: backup[name] for name in aiand_entries} + + +def test_aiand_models_listed_by_provider(monkeypatch: pytest.MonkeyPatch) -> None: + package_root = Path(litellm.__file__).parent + backup = json.loads((package_root / "model_prices_and_context_window_backup.json").read_text()) + aiand_keys = {name for name in backup if name.startswith("aiand/")} + assert aiand_keys + + monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") + monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) + monkeypatch.setattr(litellm, "models_by_provider", dict(litellm.models_by_provider)) + litellm.add_known_models() + + assert "aiand" in litellm.models_by_provider + assert set(litellm.models_by_provider["aiand"]) == aiand_keys + assert set(litellm.get_valid_models(custom_llm_provider="aiand")) == aiand_keys + + +def test_aiand_is_available_in_add_model_form() -> None: + fields_path = Path(litellm.__file__).parent / "proxy" / "public_endpoints" / "provider_create_fields.json" + providers = json.loads(fields_path.read_text()) + aiand = next(provider for provider in providers if provider["litellm_provider"] == "aiand") + + assert aiand["provider"] == "AIAND" + assert aiand["provider_display_name"] == "ai&" + assert aiand["default_model_placeholder"] == "aiand/deepseek-ai/deepseek-v4.1-flash" + assert {field["key"]: field["required"] for field in aiand["credential_fields"]} == { + "api_base": False, + "api_key": True, + } + + +def test_aiand_supported_endpoints() -> None: + matrix_path = Path(litellm.__file__).parent / "provider_endpoints_support_backup.json" + providers = json.loads(matrix_path.read_text())["providers"] + + assert providers["aiand"]["endpoints"] == { + "chat_completions": True, + "messages": True, + "responses": True, + "embeddings": False, + "image_generations": False, + "audio_transcriptions": False, + "audio_speech": False, + "moderations": False, + "batches": False, + "rerank": False, + "a2a": False, + "interactions": False, + } + + +def test_aiand_registered_for_text_completion() -> None: + assert "aiand" in litellm.openai_text_completion_compatible_providers + + +def test_aiand_provider_declares_completions_endpoint() -> None: + from litellm.llms.openai_like.json_loader import JSONProviderRegistry + + provider_config: Final = JSONProviderRegistry.get("aiand") + + assert provider_config is not None + assert "/v1/completions" in provider_config.supported_endpoints + + +def test_aiand_chat_completion_request() -> None: + with respx.mock() as upstream: + route: Final = upstream.post("https://api.aiand.com/v1/chat/completions").respond( + 200, + json={ + "id": "chatcmpl_aiand", + "object": "chat.completion", + "created": 1_789_550_000, + "model": "deepseek-ai/deepseek-v4.1-flash", + "choices": [ + { + "index": 0, + "message": {"role": "assistant", "content": "Hello from aiand"}, + "finish_reason": "stop", + } + ], + "usage": {"prompt_tokens": 4, "completion_tokens": 3, "total_tokens": 7}, + }, + ) + response: Final = litellm.completion( + model="aiand/deepseek-ai/deepseek-v4.1-flash", + messages=[{"role": "user", "content": "Say hello"}], + api_key="aiand-test-key", + ) + + request: Final = route.calls.last.request + body: Final = json.loads(request.content) + assert route.call_count == 1 + assert str(request.url) == "https://api.aiand.com/v1/chat/completions" + assert request.headers["authorization"] == "Bearer aiand-test-key" + assert body["model"] == "deepseek-ai/deepseek-v4.1-flash" + assert body["messages"] == [{"role": "user", "content": "Say hello"}] + assert response.choices[0].message.content == "Hello from aiand" + + +def test_aiand_responses_request() -> None: + with respx.mock() as upstream: + route: Final = upstream.post("https://api.aiand.com/v1/responses").respond( + 200, + json={ + "id": "resp_aiand", + "object": "response", + "created_at": 1_789_550_000, + "model": "deepseek-ai/deepseek-v4.1-flash", + "status": "completed", + "output": [ + { + "id": "msg_aiand", + "type": "message", + "role": "assistant", + "status": "completed", + "content": [{"type": "output_text", "text": "Hello from aiand", "annotations": []}], + } + ], + "usage": {"input_tokens": 4, "output_tokens": 3, "total_tokens": 7}, + }, + ) + response: Final = litellm.responses( + model="aiand/deepseek-ai/deepseek-v4.1-flash", + input="Say hello", + api_key="aiand-test-key", + ) + + request: Final = route.calls.last.request + body: Final = json.loads(request.content) + assert route.call_count == 1 + assert str(request.url) == "https://api.aiand.com/v1/responses" + assert request.headers["authorization"] == "Bearer aiand-test-key" + assert body["model"] == "deepseek-ai/deepseek-v4.1-flash" + assert body["input"] == "Say hello" + assert response.output[0].content[0].text == "Hello from aiand" + + +def test_aiand_text_completion_request(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("AIAND_API_KEY", "aiand-test-key") + + with respx.mock() as upstream: + route: Final = upstream.post("https://api.aiand.com/v1/completions").respond( + 200, + json={ + "id": "cmpl_aiand", + "object": "text_completion", + "created": 1_789_550_000, + "model": "deepseek-ai/deepseek-v4.1-flash", + "choices": [ + { + "text": "Hello from aiand", + "index": 0, + "logprobs": None, + "finish_reason": "stop", + } + ], + "usage": {"prompt_tokens": 2, "completion_tokens": 4, "total_tokens": 6}, + }, + ) + response: Final = litellm.text_completion( + model="aiand/deepseek-ai/deepseek-v4.1-flash", + prompt="Say hello", + ) + + request: Final = route.calls.last.request + body: Final = json.loads(request.content) + assert route.call_count == 1 + assert str(request.url) == "https://api.aiand.com/v1/completions" + assert request.headers["authorization"] == "Bearer aiand-test-key" + assert body["model"] == "deepseek-ai/deepseek-v4.1-flash" + assert body["prompt"] == "Say hello" + assert response.object == "text_completion" + assert response.choices[0].text == "Hello from aiand" + + +@pytest.mark.asyncio +async def test_aiand_anthropic_messages_request(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setattr(litellm, "disable_aiohttp_transport", True) + monkeypatch.setattr(litellm, "in_memory_llm_clients_cache", LLMClientCache()) + with respx.mock() as upstream: + route: Final = upstream.post("https://api.aiand.com/v1/messages").respond( + 200, + json={ + "id": "msg_aiand", + "type": "message", + "role": "assistant", + "model": "deepseek-ai/deepseek-v4.1-flash", + "content": [{"type": "text", "text": "Hello from aiand"}], + "stop_reason": "end_turn", + "stop_sequence": None, + "usage": {"input_tokens": 4, "output_tokens": 3}, + }, + ) + response: Final = await litellm.anthropic.messages.acreate( + model="aiand/deepseek-ai/deepseek-v4.1-flash", + messages=[{"role": "user", "content": "Say hello"}], + max_tokens=32, + api_key="aiand-test-key", + ) + + request: Final = route.calls.last.request + body: Final = json.loads(request.content) + assert route.call_count == 1 + assert str(request.url) == "https://api.aiand.com/v1/messages" + assert request.headers["authorization"] == "Bearer aiand-test-key" + assert request.headers["anthropic-version"] == "2023-06-01" + assert body["model"] == "deepseek-ai/deepseek-v4.1-flash" + assert body["messages"] == [{"role": "user", "content": "Say hello"}] + assert response["content"][0]["text"] == "Hello from aiand" diff --git a/tests/unit/test_sync_aiand_models.py b/tests/unit/test_sync_aiand_models.py new file mode 100644 index 00000000000..d3ae4497d39 --- /dev/null +++ b/tests/unit/test_sync_aiand_models.py @@ -0,0 +1,269 @@ +import importlib.util +import json +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[2] +SCRIPT = ROOT / "scripts" / "sync_aiand_models.py" + +_spec = importlib.util.spec_from_file_location("sync_aiand_models", SCRIPT) +assert _spec is not None and _spec.loader is not None +sync = importlib.util.module_from_spec(_spec) +_spec.loader.exec_module(sync) + + +def _model(**overrides: object) -> dict[str, object]: + model: dict[str, object] = { + "id": "acme/chat-1", + "name": "Chat 1", + "family": "chat", + "reasoning": False, + "tool_call": True, + "structured_output": True, + "temperature": True, + "attachment": False, + "open_weights": False, + "cost": {"input": 1.0, "output": 2.0, "cache_read": 0.5}, + "limit": {"context": 8192, "output": 1024}, + "modalities": {"input": ["text"]}, + } + model.update(overrides) + return model + + +def _spec_json(*models: dict[str, object]) -> bytes: + payload: dict[str, object] = {"aiand": {"models": {model["id"]: model for model in models}}} + return json.dumps(payload).encode() + + +@pytest.mark.parametrize( + ("per_million", "expected"), + [ + (3, 3e-06), + (15, 1.5e-05), + (1.4, 1.4e-06), + (0.25999999999999995, 2.6e-07), + (0.060000000000000005, 6e-08), + (1.0399999999999998, 1.04e-06), + (0, 0.0), + ], +) +def test_per_token_normalizes_float_artifacts(per_million: float, expected: float) -> None: + assert sync.per_token(per_million) == expected + + +def test_load_spec_raises_on_shape_change() -> None: + with pytest.raises(sync.SyncError): + sync.load_spec(b'{"aiand": {"models": [{"id": "x"}]}}') + + +def test_load_spec_raises_when_no_models_remain() -> None: + with pytest.raises(sync.SyncError): + sync.load_spec(b'{"aiand": {"models": {}}}') + + +def test_load_spec_raises_when_id_mismatches_key() -> None: + raw = json.dumps({"aiand": {"models": {"acme/chat-1": _model(id="acme/other")}}}).encode() + with pytest.raises(sync.SyncError): + sync.load_spec(raw) + + +def test_added_model_lands_in_cost_map_with_expected_fields() -> None: + spec = sync.load_spec(_spec_json(_model(reasoning_options=[{"type": "effort", "values": ["low", "high", "max"]}]))) + outcome = sync.compute_sync({}, spec) + entry = outcome.cost_map["aiand/acme/chat-1"] + assert entry["litellm_provider"] == "aiand" + assert entry["mode"] == "chat" + assert entry["input_cost_per_token"] == 1e-06 + assert entry["output_cost_per_token"] == 2e-06 + assert entry["cache_read_input_token_cost"] == 5e-07 + assert entry["max_input_tokens"] == 8192 + assert entry["max_output_tokens"] == 1024 + assert entry["max_tokens"] == 1024 + assert entry["supports_function_calling"] is True + assert entry["supports_parallel_function_calling"] is True + assert entry["supports_tool_choice"] is True + assert entry["supports_response_schema"] is True + assert entry["supports_reasoning"] is False + assert entry["reasoning_effort_levels"] == ["low", "high", "max"] + assert entry["supports_vision"] is False + assert entry["source"] == "https://api.aiand.com/v1/api.json" + assert entry["supported_endpoints"] == ["/v1/chat/completions", "/v1/responses", "/v1/messages"] + assert outcome.added == ("aiand/acme/chat-1",) + assert outcome.has_changes is True + + +def test_updated_price_is_detected_and_rendered() -> None: + baseline = sync.compute_sync({}, sync.load_spec(_spec_json(_model()))) + changed = _model(cost={"input": 3.0, "output": 2.0, "cache_read": 0.5}) + outcome = sync.compute_sync(baseline.cost_map, sync.load_spec(_spec_json(changed))) + assert outcome.added == () + assert len(outcome.updated) == 1 + assert outcome.updated[0].startswith("aiand/acme/chat-1:") + assert "input_cost_per_token" in outcome.updated[0] + assert outcome.cost_map["aiand/acme/chat-1"]["input_cost_per_token"] == 3e-06 + assert outcome.has_changes is True + + +def test_dropped_effort_option_is_reported_and_cleared() -> None: + offered = _model(reasoning_options=[{"type": "effort", "values": ["low", "high"]}]) + baseline = sync.compute_sync({}, sync.load_spec(_spec_json(offered))) + outcome = sync.compute_sync(baseline.cost_map, sync.load_spec(_spec_json(_model()))) + assert outcome.added == () + assert len(outcome.updated) == 1 + assert outcome.updated[0].startswith("aiand/acme/chat-1:") + assert "reasoning_effort_levels" in outcome.updated[0] + assert outcome.cost_map["aiand/acme/chat-1"]["reasoning_effort_levels"] == [] + assert outcome.has_changes is True + + +def test_removed_model_is_stamped_and_counted_as_updated( + monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setattr(sync, "_today", lambda: "2026-10-03") + baseline = sync.compute_sync({}, sync.load_spec(_spec_json(_model(), _model(id="acme/chat-2", name="Chat 2")))) + remaining = sync.load_spec(_spec_json(_model())) + outcome = sync.compute_sync(baseline.cost_map, remaining) + assert outcome.added == () + assert outcome.removed == ("aiand/acme/chat-2",) + assert len(outcome.updated) == 1 + assert outcome.updated[0].startswith("aiand/acme/chat-2:") + assert "absent_from_spec_since" in outcome.updated[0] + assert outcome.cost_map["aiand/acme/chat-2"]["metadata"] == {"absent_from_spec_since": "2026-10-03"} + assert any("aiand/acme/chat-2" in warning and "human" in warning for warning in outcome.warnings) + assert outcome.has_changes is True + + +def test_already_stamped_absent_model_keeps_the_earliest_date( + monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setattr(sync, "_today", lambda: "2026-10-03") + baseline = sync.compute_sync({}, sync.load_spec(_spec_json(_model(), _model(id="acme/chat-2", name="Chat 2")))) + stamped_entry = dict(baseline.cost_map["aiand/acme/chat-2"]) + stamped_entry["metadata"] = {"absent_from_spec_since": "2026-09-01"} + registry = {**baseline.cost_map, "aiand/acme/chat-2": stamped_entry} + outcome = sync.compute_sync(registry, sync.load_spec(_spec_json(_model()))) + assert outcome.removed == ("aiand/acme/chat-2",) + assert outcome.updated == () + assert outcome.cost_map["aiand/acme/chat-2"]["metadata"] == {"absent_from_spec_since": "2026-09-01"} + assert outcome.has_changes is True + + +def test_reappeared_model_clears_the_stamp_and_counts_as_updated() -> None: + spec = sync.load_spec(_spec_json(_model())) + baseline = sync.compute_sync({}, spec).cost_map + stamped_entry = dict(baseline["aiand/acme/chat-1"]) + stamped_entry["metadata"] = {"absent_from_spec_since": "2026-09-01"} + registry = {**baseline, "aiand/acme/chat-1": stamped_entry} + outcome = sync.compute_sync(registry, spec) + assert outcome.added == () + assert outcome.removed == () + assert len(outcome.updated) == 1 + assert outcome.updated[0].startswith("aiand/acme/chat-1:") + assert "absent_from_spec_since" in outcome.updated[0] + assert "metadata" not in outcome.cost_map["aiand/acme/chat-1"] + assert outcome.has_changes is True + + +def test_updated_entry_preserves_the_absence_marker_as_an_extra() -> None: + model = sync.load_spec(_spec_json(_model()))["acme/chat-1"] + entry = {**sync._new_entry(model), "metadata": {"absent_from_spec_since": "2026-09-01"}} + new_entry, changes = sync._updated_entry(entry, model) + assert new_entry["metadata"] == {"absent_from_spec_since": "2026-09-01"} + assert changes == () + + +def test_parallel_function_calling_follows_tool_call() -> None: + spec = sync.load_spec(_spec_json(_model(tool_call=False))) + outcome = sync.compute_sync({}, spec) + entry = outcome.cost_map["aiand/acme/chat-1"] + assert entry["supports_function_calling"] is False + assert entry["supports_parallel_function_calling"] is False + assert entry["supports_tool_choice"] is False + + +def test_new_keys_land_at_the_end_of_the_provider_block() -> None: + registry = { + "aaa": {}, + "aiand/acme/chat-1": {}, + "zzz": {}, + } + spec = sync.load_spec(_spec_json(_model(), _model(id="acme/new", name="New"))) + outcome = sync.compute_sync(registry, spec) + assert list(outcome.cost_map) == ["aaa", "aiand/acme/chat-1", "aiand/acme/new", "zzz"] + + +def test_pr_body_renders_none_placeholders_for_empty_sections() -> None: + outcome = sync.compute_sync({}, sync.load_spec(_spec_json(_model()))) + body = sync.render_pr_body(outcome) + assert "### Added (1)" in body + assert "### Updated (0)\n- none" in body + assert "### Removed from the spec (0)\n- none" in body + assert sync.render_summary(outcome) == "added=1 updated=0 removed=0 warnings=0" + + +def test_write_updates_root_and_backup_maps_identically(tmp_path: Path) -> None: + repo_root = tmp_path + for relpath in sync.COST_MAP_RELPATHS: + target = repo_root / relpath + target.parent.mkdir(parents=True, exist_ok=True) + target.write_text("{}\n") + spec_path = repo_root / "spec.json" + spec_path.write_bytes(_spec_json(_model())) + pr_body = repo_root / "pr_body.md" + exit_code = sync.main( + [ + "--write", + "--spec-json", + str(spec_path), + "--pr-body-file", + str(pr_body), + "--repo-root", + str(repo_root), + ] + ) + assert exit_code == 0 + root_map = json.loads((repo_root / sync.COST_MAP_RELPATHS[0]).read_text()) + backup_map = json.loads((repo_root / sync.COST_MAP_RELPATHS[1]).read_text()) + assert root_map == backup_map + assert "aiand/acme/chat-1" in root_map + + +def test_removal_only_sync_stamps_and_writes_files( + tmp_path: Path, capsys: pytest.CaptureFixture[str], monkeypatch: pytest.MonkeyPatch +) -> None: + monkeypatch.setattr(sync, "_today", lambda: "2026-10-03") + repo_root = tmp_path + spec = sync.load_spec(_spec_json(_model(), _model(id="acme/chat-2", name="Chat 2"))) + registry = sync.compute_sync({}, spec).cost_map + for relpath in sync.COST_MAP_RELPATHS: + target = repo_root / relpath + target.parent.mkdir(parents=True, exist_ok=True) + target.write_text(json.dumps(registry, indent=4) + "\n") + remaining = repo_root / "remaining.json" + remaining.write_bytes(_spec_json(_model())) + pr_body = repo_root / "pr_body.md" + exit_code = sync.main( + [ + "--write", + "--spec-json", + str(remaining), + "--pr-body-file", + str(pr_body), + "--repo-root", + str(repo_root), + ] + ) + assert exit_code == 0 + assert "updated=1 removed=1 warnings=1" in capsys.readouterr().out + body = pr_body.read_text() + assert "### Added (0)" in body + assert "### Updated (1)" in body + assert "### Removed from the spec (1)" in body + assert "`aiand/acme/chat-2`" in body + assert "### Warnings needing a human call" not in body + root_map = json.loads((repo_root / sync.COST_MAP_RELPATHS[0]).read_text()) + backup_map = json.loads((repo_root / sync.COST_MAP_RELPATHS[1]).read_text()) + assert root_map == backup_map + assert root_map["aiand/acme/chat-2"]["metadata"] == {"absent_from_spec_since": "2026-10-03"} diff --git a/ui/litellm-dashboard/public/assets/logos/aiand.svg b/ui/litellm-dashboard/public/assets/logos/aiand.svg new file mode 100644 index 00000000000..da00acc02ac --- /dev/null +++ b/ui/litellm-dashboard/public/assets/logos/aiand.svg @@ -0,0 +1 @@ +ai&ai& diff --git a/ui/litellm-dashboard/src/components/provider_info_helpers.test.tsx b/ui/litellm-dashboard/src/components/provider_info_helpers.test.tsx index 7cfdaf3275d..91d14ff20e6 100644 --- a/ui/litellm-dashboard/src/components/provider_info_helpers.test.tsx +++ b/ui/litellm-dashboard/src/components/provider_info_helpers.test.tsx @@ -73,6 +73,19 @@ describe("provider_info_helpers", () => { expect(fromEnumKey.logo).toBe(providerLogoMap[Providers.SCX_AI]); }); + it("should map aiand slug to the ai& display name and logo", () => { + const fromSlug = getProviderLogoAndName("aiand"); + expect(fromSlug.displayName).toBe(Providers.AIAND); + expect(fromSlug.logo).toBe(providerLogoMap[Providers.AIAND]); + expect(fromSlug.logo).toBeTruthy(); + }); + + it("should map AIAND enum key to the ai& display name and logo", () => { + const fromEnumKey = getProviderLogoAndName("AIAND"); + expect(fromEnumKey.displayName).toBe(Providers.AIAND); + expect(fromEnumKey.logo).toBe(providerLogoMap[Providers.AIAND]); + }); + it("should map bedrock_mantle slug to Bedrock Mantle display name and logo", () => { const result = getProviderLogoAndName("bedrock_mantle"); expect(result.displayName).toBe(Providers.BedrockMantle); @@ -229,6 +242,10 @@ describe("provider_info_helpers", () => { expect(getPlaceholder(Providers.SCX_AI)).toBe("scx-ai/GLM-5.2"); }); + it("should return an aiand model placeholder for AIAND provider", () => { + expect(getPlaceholder(Providers.AIAND)).toBe("aiand/deepseek-ai/deepseek-v4.1-flash"); + }); + it("should return an edenai model placeholder for EDENAI provider", () => { expect(getPlaceholder(Providers.EDENAI)).toBe("edenai/openai/gpt-mini-latest"); }); diff --git a/ui/litellm-dashboard/src/components/provider_info_helpers.tsx b/ui/litellm-dashboard/src/components/provider_info_helpers.tsx index 5ea693bea10..dcb450b355f 100644 --- a/ui/litellm-dashboard/src/components/provider_info_helpers.tsx +++ b/ui/litellm-dashboard/src/components/provider_info_helpers.tsx @@ -53,6 +53,7 @@ import runwayLogo from "../../public/assets/logos/runway.png"; import sambanovaLogo from "../../public/assets/logos/sambanova.svg"; import sapLogo from "../../public/assets/logos/sap.png"; import scxAiLogo from "../../public/assets/logos/scx_ai.svg"; +import aiandLogo from "../../public/assets/logos/aiand.svg"; import snowflakeLogo from "../../public/assets/logos/snowflake.svg"; import sonioxLogo from "../../public/assets/logos/soniox.svg"; import tencentLogo from "../../public/assets/logos/tencent.svg"; @@ -165,6 +166,7 @@ export enum Providers { Sambanova = "Sambanova", SAP = "SAP Generative AI Hub", SCX_AI = "SCX.ai", + AIAND = "ai&", Snowflake = "Snowflake", Soniox = "Soniox", TEXT_COMPLETION_CODESTRAL = "Text-Completion-Codestral", @@ -285,6 +287,7 @@ export const provider_map: Record = { Sambanova: "sambanova", SAP: "sap", SCX_AI: "scx-ai", + AIAND: "aiand", Snowflake: "snowflake", Soniox: "soniox", TEXT_COMPLETION_CODESTRAL: "text-completion-codestral", @@ -385,6 +388,7 @@ export const providerLogoMap: Partial> = { [Providers.Sambanova]: sambanovaLogo.src, [Providers.SAP]: sapLogo.src, [Providers.SCX_AI]: scxAiLogo.src, + [Providers.AIAND]: aiandLogo.src, [Providers.Snowflake]: snowflakeLogo.src, [Providers.Soniox]: sonioxLogo.src, [Providers.Tencent]: tencentLogo.src, @@ -456,6 +460,7 @@ const providerPlaceholderMap: Partial> = { [Providers.SageMaker]: "sagemaker/jumpstart-dft-meta-textgeneration-llama-2-7b", [Providers.Sail]: "sail/openai/gpt-oss-120b", [Providers.SCX_AI]: "scx-ai/GLM-5.2", + [Providers.AIAND]: "aiand/deepseek-ai/deepseek-v4.1-flash", [Providers.Snowflake]: "snowflake/mistral-7b", [Providers.Tencent]: "tencent/deepseek-v4-pro", [Providers.Vertex_AI]: "gemini-pro",