feat(mistral): register voxtral models in the cost map

Voxtral was absent from the cost map, so mistral/* on the proxy expanded to a model
list without it and /v1/models never surfaced the audio models.

Adds voxtral-mini-2602 / voxtral-mini-latest as mode=audio_transcription priced at
$0.003/min against /v1/audio/transcriptions, and voxtral-small-2507 /
voxtral-small-latest as an audio-capable chat model at $0.1/$0.3 per MTok. Names,
aliases, context windows and capabilities come from a live GET https://api.mistral.ai/v1/models;
the realtime transcription and TTS voxtral variants are left out since LiteLLM has no
route for them yet.
This commit is contained in:
Devin AI 2026-07-25 13:08:43 +00:00
parent b9b27c2beb
commit cba2056b6a
3 changed files with 171 additions and 0 deletions

View file

@ -28066,6 +28066,52 @@
"supports_tool_choice": true,
"supports_vision": true
},
"mistral/voxtral-mini-2602": {
"input_cost_per_second": 5e-05,
"litellm_provider": "mistral",
"max_input_tokens": 16384,
"mode": "audio_transcription",
"source": "https://docs.mistral.ai/models/model-cards/voxtral-mini-transcribe-26-02",
"supported_endpoints": [
"/v1/audio/transcriptions"
]
},
"mistral/voxtral-mini-latest": {
"input_cost_per_second": 5e-05,
"litellm_provider": "mistral",
"max_input_tokens": 16384,
"mode": "audio_transcription",
"source": "https://docs.mistral.ai/models/model-cards/voxtral-mini-transcribe-26-02",
"supported_endpoints": [
"/v1/audio/transcriptions"
]
},
"mistral/voxtral-small-2507": {
"input_cost_per_token": 1e-07,
"litellm_provider": "mistral",
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"max_tokens": 32768,
"mode": "chat",
"output_cost_per_token": 3e-07,
"source": "https://docs.mistral.ai/models/model-cards/voxtral-small-25-07",
"supports_audio_input": true,
"supports_function_calling": true,
"supports_tool_choice": true
},
"mistral/voxtral-small-latest": {
"input_cost_per_token": 1e-07,
"litellm_provider": "mistral",
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"max_tokens": 32768,
"mode": "chat",
"output_cost_per_token": 3e-07,
"source": "https://docs.mistral.ai/models/model-cards/voxtral-small-25-07",
"supports_audio_input": true,
"supports_function_calling": true,
"supports_tool_choice": true
},
"moonshot.kimi-k2-thinking": {
"input_cost_per_token": 6e-07,
"litellm_provider": "bedrock_converse",

View file

@ -28141,6 +28141,52 @@
"supports_tool_choice": true,
"supports_vision": true
},
"mistral/voxtral-mini-2602": {
"input_cost_per_second": 5e-05,
"litellm_provider": "mistral",
"max_input_tokens": 16384,
"mode": "audio_transcription",
"source": "https://docs.mistral.ai/models/model-cards/voxtral-mini-transcribe-26-02",
"supported_endpoints": [
"/v1/audio/transcriptions"
]
},
"mistral/voxtral-mini-latest": {
"input_cost_per_second": 5e-05,
"litellm_provider": "mistral",
"max_input_tokens": 16384,
"mode": "audio_transcription",
"source": "https://docs.mistral.ai/models/model-cards/voxtral-mini-transcribe-26-02",
"supported_endpoints": [
"/v1/audio/transcriptions"
]
},
"mistral/voxtral-small-2507": {
"input_cost_per_token": 1e-07,
"litellm_provider": "mistral",
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"max_tokens": 32768,
"mode": "chat",
"output_cost_per_token": 3e-07,
"source": "https://docs.mistral.ai/models/model-cards/voxtral-small-25-07",
"supports_audio_input": true,
"supports_function_calling": true,
"supports_tool_choice": true
},
"mistral/voxtral-small-latest": {
"input_cost_per_token": 1e-07,
"litellm_provider": "mistral",
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"max_tokens": 32768,
"mode": "chat",
"output_cost_per_token": 3e-07,
"source": "https://docs.mistral.ai/models/model-cards/voxtral-small-25-07",
"supports_audio_input": true,
"supports_function_calling": true,
"supports_tool_choice": true
},
"moonshot.kimi-k2-thinking": {
"input_cost_per_token": 6e-07,
"litellm_provider": "bedrock_converse",

View file

@ -0,0 +1,79 @@
import json
from pathlib import Path
import pytest
import litellm
from litellm.proxy.auth.model_checks import get_provider_models
from litellm.types.utils import TranscriptionResponse
TRANSCRIPTION_MODELS = ("mistral/voxtral-mini-2602", "mistral/voxtral-mini-latest")
CHAT_MODELS = ("mistral/voxtral-small-2507", "mistral/voxtral-small-latest")
ALL_VOXTRAL_MODELS = TRANSCRIPTION_MODELS + CHAT_MODELS
def _load_cost_maps() -> tuple[dict, dict]:
repo_root = Path(__file__).parents[2]
with open(repo_root / "model_prices_and_context_window.json") as f:
main_cost = json.load(f)
with open(repo_root / "litellm" / "model_prices_and_context_window_backup.json") as f:
backup_cost = json.load(f)
return main_cost, backup_cost
@pytest.fixture
def local_cost_map(monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
local_map = litellm.get_model_cost_map(url="")
monkeypatch.setattr(litellm, "model_cost", local_map)
litellm.add_known_models(local_map)
litellm.get_model_info.cache_clear()
yield
litellm.get_model_info.cache_clear()
@pytest.mark.parametrize("model", TRANSCRIPTION_MODELS)
def test_voxtral_transcription_models_are_priced_per_second(model: str) -> None:
main_cost, _ = _load_cost_maps()
info = main_cost[model]
assert info["mode"] == "audio_transcription"
assert info["supported_endpoints"] == ["/v1/audio/transcriptions"]
assert info["input_cost_per_second"] == pytest.approx(0.003 / 60)
assert "output_cost_per_second" not in info
@pytest.mark.parametrize("model", CHAT_MODELS)
def test_voxtral_small_is_an_audio_capable_chat_model(model: str) -> None:
main_cost, _ = _load_cost_maps()
info = main_cost[model]
assert info["mode"] == "chat"
assert info["supports_audio_input"] is True
assert info["input_cost_per_token"] == pytest.approx(1e-07)
assert info["output_cost_per_token"] == pytest.approx(3e-07)
@pytest.mark.parametrize("model", ALL_VOXTRAL_MODELS)
def test_backup_cost_map_matches_main(model: str) -> None:
main_cost, backup_cost = _load_cost_maps()
assert backup_cost.get(model) == main_cost.get(model)
def test_voxtral_models_are_discoverable_for_the_mistral_wildcard(local_cost_map: None) -> None:
"""Regression for #34616: `mistral/*` on the proxy expands to the provider's known models,
so voxtral was missing from /v1/models until it landed in the cost map."""
provider_models = get_provider_models(provider="mistral") or []
assert set(ALL_VOXTRAL_MODELS).issubset(set(provider_models))
def test_voxtral_transcription_cost_is_billed_per_audio_second(local_cost_map: None) -> None:
response = TranscriptionResponse(text="demo text")
response.duration = 90.0
cost = litellm.completion_cost(
completion_response=response,
model="mistral/voxtral-mini-latest",
custom_llm_provider="mistral",
call_type="atranscription",
)
assert cost == pytest.approx(90.0 * 0.003 / 60)