diff --git a/docs/my-website/docs/providers/hpc_ai.md b/docs/my-website/docs/providers/hpc_ai.md index 4be00e9a7f4..79389d4d396 100644 --- a/docs/my-website/docs/providers/hpc_ai.md +++ b/docs/my-website/docs/providers/hpc_ai.md @@ -1,6 +1,6 @@ # HPC-AI -[HPC-AI](https://api.hpc-ai.com) provides an OpenAI-compatible inference API at `https://api.hpc-ai.com/inference/v1`. +[HPC-AI](https://www.hpc-ai.com/) provides an OpenAI-compatible inference API at `https://api.hpc-ai.com/inference/v1`. :::tip @@ -22,8 +22,6 @@ Optional: override the base URL (defaults to `https://api.hpc-ai.com/inference/v os.environ["HPC_AI_API_BASE"] = "https://api.hpc-ai.com/inference/v1" ``` -If you use another env name such as `HPC_AI_BASE_URL`, map it to `api_base` in your LiteLLM call or proxy `litellm_params`; LiteLLM reads `HPC_AI_API_BASE` by default. - ## Sample Usage: Chat completion ```python @@ -76,11 +74,18 @@ litellm --config /path/to/config.yaml 3. Send requests to the proxy using your alias (`hpc-ai-minimax` in the example above). -## Supported models (examples) +## Supported models | LiteLLM model id | Notes | | ---------------- | ----- | | `hpc_ai/minimax/minimax-m2.5` | MiniMax M2.5 | | `hpc_ai/moonshotai/kimi-k2.5` | Kimi K2.5 | -Pricing in `model_prices_and_context_window.json` may use placeholder token costs; set real rates when your billing API is available. +## Pricing + +Costs in `model_prices_and_context_window.json` use USD per token. Approximate list rates (per 1M tokens): + +| Model | Uncached input | Cached input | Output | +| ----- | -------------- | ------------ | ------ | +| M2.5 (`minimax/minimax-m2.5`) | $0.30 | $0.03 | $1.20 | +| K2.5 (`moonshotai/kimi-k2.5`) | $0.45 | $0.07 | $2.25 | diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 1758ea01481..fdac1253cb3 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -20189,26 +20189,30 @@ ] }, "hpc_ai/minimax/minimax-m2.5": { - "max_tokens": 262144, - "max_input_tokens": 262144, - "max_output_tokens": 262144, - "input_cost_per_token": 0.0, - "output_cost_per_token": 0.0, + "cache_read_input_token_cost": 3e-08, + "input_cost_per_token": 3e-07, "litellm_provider": "hpc_ai", + "max_input_tokens": 204800, + "max_output_tokens": 204800, + "max_tokens": 204800, "mode": "chat", + "output_cost_per_token": 1.2e-06, + "source": "https://api.hpc-ai.com/inference/v1", "supports_function_calling": true, - "source": "https://api.hpc-ai.com/inference/v1" + "supports_prompt_caching": true }, "hpc_ai/moonshotai/kimi-k2.5": { - "max_tokens": 262144, + "cache_read_input_token_cost": 7e-08, + "input_cost_per_token": 4.5e-07, + "litellm_provider": "hpc_ai", "max_input_tokens": 262144, "max_output_tokens": 262144, - "input_cost_per_token": 0.0, - "output_cost_per_token": 0.0, - "litellm_provider": "hpc_ai", + "max_tokens": 262144, "mode": "chat", + "output_cost_per_token": 2.25e-06, + "source": "https://api.hpc-ai.com/inference/v1", "supports_function_calling": true, - "source": "https://api.hpc-ai.com/inference/v1" + "supports_prompt_caching": true }, "hyperbolic/NousResearch/Hermes-3-Llama-3.1-70B": { "input_cost_per_token": 1.2e-07,