fix some empty config

This commit is contained in:
lioZ129 2026-03-26 14:34:44 +08:00
parent f0ea81acce
commit bf12f7d947
2 changed files with 25 additions and 16 deletions

View file

@ -1,6 +1,6 @@
# HPC-AI
[HPC-AI](https://api.hpc-ai.com) provides an OpenAI-compatible inference API at `https://api.hpc-ai.com/inference/v1`.
[HPC-AI](https://www.hpc-ai.com/) provides an OpenAI-compatible inference API at `https://api.hpc-ai.com/inference/v1`.
:::tip
@ -22,8 +22,6 @@ Optional: override the base URL (defaults to `https://api.hpc-ai.com/inference/v
os.environ["HPC_AI_API_BASE"] = "https://api.hpc-ai.com/inference/v1"
```
If you use another env name such as `HPC_AI_BASE_URL`, map it to `api_base` in your LiteLLM call or proxy `litellm_params`; LiteLLM reads `HPC_AI_API_BASE` by default.
## Sample Usage: Chat completion
```python
@ -76,11 +74,18 @@ litellm --config /path/to/config.yaml
3. Send requests to the proxy using your alias (`hpc-ai-minimax` in the example above).
## Supported models (examples)
## Supported models
| LiteLLM model id | Notes |
| ---------------- | ----- |
| `hpc_ai/minimax/minimax-m2.5` | MiniMax M2.5 |
| `hpc_ai/moonshotai/kimi-k2.5` | Kimi K2.5 |
Pricing in `model_prices_and_context_window.json` may use placeholder token costs; set real rates when your billing API is available.
## Pricing
Costs in `model_prices_and_context_window.json` use USD per token. Approximate list rates (per 1M tokens):
| Model | Uncached input | Cached input | Output |
| ----- | -------------- | ------------ | ------ |
| M2.5 (`minimax/minimax-m2.5`) | $0.30 | $0.03 | $1.20 |
| K2.5 (`moonshotai/kimi-k2.5`) | $0.45 | $0.07 | $2.25 |

View file

@ -20189,26 +20189,30 @@
]
},
"hpc_ai/minimax/minimax-m2.5": {
"max_tokens": 262144,
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"input_cost_per_token": 0.0,
"output_cost_per_token": 0.0,
"cache_read_input_token_cost": 3e-08,
"input_cost_per_token": 3e-07,
"litellm_provider": "hpc_ai",
"max_input_tokens": 204800,
"max_output_tokens": 204800,
"max_tokens": 204800,
"mode": "chat",
"output_cost_per_token": 1.2e-06,
"source": "https://api.hpc-ai.com/inference/v1",
"supports_function_calling": true,
"source": "https://api.hpc-ai.com/inference/v1"
"supports_prompt_caching": true
},
"hpc_ai/moonshotai/kimi-k2.5": {
"max_tokens": 262144,
"cache_read_input_token_cost": 7e-08,
"input_cost_per_token": 4.5e-07,
"litellm_provider": "hpc_ai",
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"input_cost_per_token": 0.0,
"output_cost_per_token": 0.0,
"litellm_provider": "hpc_ai",
"max_tokens": 262144,
"mode": "chat",
"output_cost_per_token": 2.25e-06,
"source": "https://api.hpc-ai.com/inference/v1",
"supports_function_calling": true,
"source": "https://api.hpc-ai.com/inference/v1"
"supports_prompt_caching": true
},
"hyperbolic/NousResearch/Hermes-3-Llama-3.1-70B": {
"input_cost_per_token": 1.2e-07,