From cba2056b6a6ce4439f76464e2db0ce6f7c42a65e Mon Sep 17 00:00:00 2001 From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Sat, 25 Jul 2026 13:08:43 +0000 Subject: [PATCH] feat(mistral): register voxtral models in the cost map Voxtral was absent from the cost map, so mistral/* on the proxy expanded to a model list without it and /v1/models never surfaced the audio models. Adds voxtral-mini-2602 / voxtral-mini-latest as mode=audio_transcription priced at $0.003/min against /v1/audio/transcriptions, and voxtral-small-2507 / voxtral-small-latest as an audio-capable chat model at $0.1/$0.3 per MTok. Names, aliases, context windows and capabilities come from a live GET https://api.mistral.ai/v1/models; the realtime transcription and TTS voxtral variants are left out since LiteLLM has no route for them yet. --- ...odel_prices_and_context_window_backup.json | 46 +++++++++++ model_prices_and_context_window.json | 46 +++++++++++ .../test_mistral_voxtral_models.py | 79 +++++++++++++++++++ 3 files changed, 171 insertions(+) create mode 100644 tests/test_litellm/test_mistral_voxtral_models.py diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index d43eda39b1f..fd4105b0c69 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -28066,6 +28066,52 @@ "supports_tool_choice": true, "supports_vision": true }, + "mistral/voxtral-mini-2602": { + "input_cost_per_second": 5e-05, + "litellm_provider": "mistral", + "max_input_tokens": 16384, + "mode": "audio_transcription", + "source": "https://docs.mistral.ai/models/model-cards/voxtral-mini-transcribe-26-02", + "supported_endpoints": [ + "/v1/audio/transcriptions" + ] + }, + "mistral/voxtral-mini-latest": { + "input_cost_per_second": 5e-05, + "litellm_provider": "mistral", + "max_input_tokens": 16384, + "mode": "audio_transcription", + "source": "https://docs.mistral.ai/models/model-cards/voxtral-mini-transcribe-26-02", + "supported_endpoints": [ + "/v1/audio/transcriptions" + ] + }, + "mistral/voxtral-small-2507": { + "input_cost_per_token": 1e-07, + "litellm_provider": "mistral", + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 3e-07, + "source": "https://docs.mistral.ai/models/model-cards/voxtral-small-25-07", + "supports_audio_input": true, + "supports_function_calling": true, + "supports_tool_choice": true + }, + "mistral/voxtral-small-latest": { + "input_cost_per_token": 1e-07, + "litellm_provider": "mistral", + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 3e-07, + "source": "https://docs.mistral.ai/models/model-cards/voxtral-small-25-07", + "supports_audio_input": true, + "supports_function_calling": true, + "supports_tool_choice": true + }, "moonshot.kimi-k2-thinking": { "input_cost_per_token": 6e-07, "litellm_provider": "bedrock_converse", diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 749b2566c2a..2aa172b8f11 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -28141,6 +28141,52 @@ "supports_tool_choice": true, "supports_vision": true }, + "mistral/voxtral-mini-2602": { + "input_cost_per_second": 5e-05, + "litellm_provider": "mistral", + "max_input_tokens": 16384, + "mode": "audio_transcription", + "source": "https://docs.mistral.ai/models/model-cards/voxtral-mini-transcribe-26-02", + "supported_endpoints": [ + "/v1/audio/transcriptions" + ] + }, + "mistral/voxtral-mini-latest": { + "input_cost_per_second": 5e-05, + "litellm_provider": "mistral", + "max_input_tokens": 16384, + "mode": "audio_transcription", + "source": "https://docs.mistral.ai/models/model-cards/voxtral-mini-transcribe-26-02", + "supported_endpoints": [ + "/v1/audio/transcriptions" + ] + }, + "mistral/voxtral-small-2507": { + "input_cost_per_token": 1e-07, + "litellm_provider": "mistral", + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 3e-07, + "source": "https://docs.mistral.ai/models/model-cards/voxtral-small-25-07", + "supports_audio_input": true, + "supports_function_calling": true, + "supports_tool_choice": true + }, + "mistral/voxtral-small-latest": { + "input_cost_per_token": 1e-07, + "litellm_provider": "mistral", + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 3e-07, + "source": "https://docs.mistral.ai/models/model-cards/voxtral-small-25-07", + "supports_audio_input": true, + "supports_function_calling": true, + "supports_tool_choice": true + }, "moonshot.kimi-k2-thinking": { "input_cost_per_token": 6e-07, "litellm_provider": "bedrock_converse", diff --git a/tests/test_litellm/test_mistral_voxtral_models.py b/tests/test_litellm/test_mistral_voxtral_models.py new file mode 100644 index 00000000000..3fafe3f4bef --- /dev/null +++ b/tests/test_litellm/test_mistral_voxtral_models.py @@ -0,0 +1,79 @@ +import json +from pathlib import Path + +import pytest + +import litellm +from litellm.proxy.auth.model_checks import get_provider_models +from litellm.types.utils import TranscriptionResponse + +TRANSCRIPTION_MODELS = ("mistral/voxtral-mini-2602", "mistral/voxtral-mini-latest") +CHAT_MODELS = ("mistral/voxtral-small-2507", "mistral/voxtral-small-latest") +ALL_VOXTRAL_MODELS = TRANSCRIPTION_MODELS + CHAT_MODELS + + +def _load_cost_maps() -> tuple[dict, dict]: + repo_root = Path(__file__).parents[2] + with open(repo_root / "model_prices_and_context_window.json") as f: + main_cost = json.load(f) + with open(repo_root / "litellm" / "model_prices_and_context_window_backup.json") as f: + backup_cost = json.load(f) + return main_cost, backup_cost + + +@pytest.fixture +def local_cost_map(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") + local_map = litellm.get_model_cost_map(url="") + monkeypatch.setattr(litellm, "model_cost", local_map) + litellm.add_known_models(local_map) + litellm.get_model_info.cache_clear() + yield + litellm.get_model_info.cache_clear() + + +@pytest.mark.parametrize("model", TRANSCRIPTION_MODELS) +def test_voxtral_transcription_models_are_priced_per_second(model: str) -> None: + main_cost, _ = _load_cost_maps() + info = main_cost[model] + assert info["mode"] == "audio_transcription" + assert info["supported_endpoints"] == ["/v1/audio/transcriptions"] + assert info["input_cost_per_second"] == pytest.approx(0.003 / 60) + assert "output_cost_per_second" not in info + + +@pytest.mark.parametrize("model", CHAT_MODELS) +def test_voxtral_small_is_an_audio_capable_chat_model(model: str) -> None: + main_cost, _ = _load_cost_maps() + info = main_cost[model] + assert info["mode"] == "chat" + assert info["supports_audio_input"] is True + assert info["input_cost_per_token"] == pytest.approx(1e-07) + assert info["output_cost_per_token"] == pytest.approx(3e-07) + + +@pytest.mark.parametrize("model", ALL_VOXTRAL_MODELS) +def test_backup_cost_map_matches_main(model: str) -> None: + main_cost, backup_cost = _load_cost_maps() + assert backup_cost.get(model) == main_cost.get(model) + + +def test_voxtral_models_are_discoverable_for_the_mistral_wildcard(local_cost_map: None) -> None: + """Regression for #34616: `mistral/*` on the proxy expands to the provider's known models, + so voxtral was missing from /v1/models until it landed in the cost map.""" + provider_models = get_provider_models(provider="mistral") or [] + assert set(ALL_VOXTRAL_MODELS).issubset(set(provider_models)) + + +def test_voxtral_transcription_cost_is_billed_per_audio_second(local_cost_map: None) -> None: + response = TranscriptionResponse(text="demo text") + response.duration = 90.0 + + cost = litellm.completion_cost( + completion_response=response, + model="mistral/voxtral-mini-latest", + custom_llm_provider="mistral", + call_type="atranscription", + ) + + assert cost == pytest.approx(90.0 * 0.003 / 60)