mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-05 02:41:56 +00:00
feat(mistral): register voxtral models in the cost map
Voxtral was absent from the cost map, so mistral/* on the proxy expanded to a model list without it and /v1/models never surfaced the audio models. Adds voxtral-mini-2602 / voxtral-mini-latest as mode=audio_transcription priced at $0.003/min against /v1/audio/transcriptions, and voxtral-small-2507 / voxtral-small-latest as an audio-capable chat model at $0.1/$0.3 per MTok. Names, aliases, context windows and capabilities come from a live GET https://api.mistral.ai/v1/models; the realtime transcription and TTS voxtral variants are left out since LiteLLM has no route for them yet.
This commit is contained in:
parent
b9b27c2beb
commit
cba2056b6a
3 changed files with 171 additions and 0 deletions
|
|
@ -28066,6 +28066,52 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true
|
||||
},
|
||||
"mistral/voxtral-mini-2602": {
|
||||
"input_cost_per_second": 5e-05,
|
||||
"litellm_provider": "mistral",
|
||||
"max_input_tokens": 16384,
|
||||
"mode": "audio_transcription",
|
||||
"source": "https://docs.mistral.ai/models/model-cards/voxtral-mini-transcribe-26-02",
|
||||
"supported_endpoints": [
|
||||
"/v1/audio/transcriptions"
|
||||
]
|
||||
},
|
||||
"mistral/voxtral-mini-latest": {
|
||||
"input_cost_per_second": 5e-05,
|
||||
"litellm_provider": "mistral",
|
||||
"max_input_tokens": 16384,
|
||||
"mode": "audio_transcription",
|
||||
"source": "https://docs.mistral.ai/models/model-cards/voxtral-mini-transcribe-26-02",
|
||||
"supported_endpoints": [
|
||||
"/v1/audio/transcriptions"
|
||||
]
|
||||
},
|
||||
"mistral/voxtral-small-2507": {
|
||||
"input_cost_per_token": 1e-07,
|
||||
"litellm_provider": "mistral",
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"max_tokens": 32768,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3e-07,
|
||||
"source": "https://docs.mistral.ai/models/model-cards/voxtral-small-25-07",
|
||||
"supports_audio_input": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"mistral/voxtral-small-latest": {
|
||||
"input_cost_per_token": 1e-07,
|
||||
"litellm_provider": "mistral",
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"max_tokens": 32768,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3e-07,
|
||||
"source": "https://docs.mistral.ai/models/model-cards/voxtral-small-25-07",
|
||||
"supports_audio_input": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"moonshot.kimi-k2-thinking": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "bedrock_converse",
|
||||
|
|
|
|||
|
|
@ -28141,6 +28141,52 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true
|
||||
},
|
||||
"mistral/voxtral-mini-2602": {
|
||||
"input_cost_per_second": 5e-05,
|
||||
"litellm_provider": "mistral",
|
||||
"max_input_tokens": 16384,
|
||||
"mode": "audio_transcription",
|
||||
"source": "https://docs.mistral.ai/models/model-cards/voxtral-mini-transcribe-26-02",
|
||||
"supported_endpoints": [
|
||||
"/v1/audio/transcriptions"
|
||||
]
|
||||
},
|
||||
"mistral/voxtral-mini-latest": {
|
||||
"input_cost_per_second": 5e-05,
|
||||
"litellm_provider": "mistral",
|
||||
"max_input_tokens": 16384,
|
||||
"mode": "audio_transcription",
|
||||
"source": "https://docs.mistral.ai/models/model-cards/voxtral-mini-transcribe-26-02",
|
||||
"supported_endpoints": [
|
||||
"/v1/audio/transcriptions"
|
||||
]
|
||||
},
|
||||
"mistral/voxtral-small-2507": {
|
||||
"input_cost_per_token": 1e-07,
|
||||
"litellm_provider": "mistral",
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"max_tokens": 32768,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3e-07,
|
||||
"source": "https://docs.mistral.ai/models/model-cards/voxtral-small-25-07",
|
||||
"supports_audio_input": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"mistral/voxtral-small-latest": {
|
||||
"input_cost_per_token": 1e-07,
|
||||
"litellm_provider": "mistral",
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"max_tokens": 32768,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3e-07,
|
||||
"source": "https://docs.mistral.ai/models/model-cards/voxtral-small-25-07",
|
||||
"supports_audio_input": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"moonshot.kimi-k2-thinking": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "bedrock_converse",
|
||||
|
|
|
|||
79
tests/test_litellm/test_mistral_voxtral_models.py
Normal file
79
tests/test_litellm/test_mistral_voxtral_models.py
Normal file
|
|
@ -0,0 +1,79 @@
|
|||
import json
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.proxy.auth.model_checks import get_provider_models
|
||||
from litellm.types.utils import TranscriptionResponse
|
||||
|
||||
TRANSCRIPTION_MODELS = ("mistral/voxtral-mini-2602", "mistral/voxtral-mini-latest")
|
||||
CHAT_MODELS = ("mistral/voxtral-small-2507", "mistral/voxtral-small-latest")
|
||||
ALL_VOXTRAL_MODELS = TRANSCRIPTION_MODELS + CHAT_MODELS
|
||||
|
||||
|
||||
def _load_cost_maps() -> tuple[dict, dict]:
|
||||
repo_root = Path(__file__).parents[2]
|
||||
with open(repo_root / "model_prices_and_context_window.json") as f:
|
||||
main_cost = json.load(f)
|
||||
with open(repo_root / "litellm" / "model_prices_and_context_window_backup.json") as f:
|
||||
backup_cost = json.load(f)
|
||||
return main_cost, backup_cost
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def local_cost_map(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
local_map = litellm.get_model_cost_map(url="")
|
||||
monkeypatch.setattr(litellm, "model_cost", local_map)
|
||||
litellm.add_known_models(local_map)
|
||||
litellm.get_model_info.cache_clear()
|
||||
yield
|
||||
litellm.get_model_info.cache_clear()
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", TRANSCRIPTION_MODELS)
|
||||
def test_voxtral_transcription_models_are_priced_per_second(model: str) -> None:
|
||||
main_cost, _ = _load_cost_maps()
|
||||
info = main_cost[model]
|
||||
assert info["mode"] == "audio_transcription"
|
||||
assert info["supported_endpoints"] == ["/v1/audio/transcriptions"]
|
||||
assert info["input_cost_per_second"] == pytest.approx(0.003 / 60)
|
||||
assert "output_cost_per_second" not in info
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", CHAT_MODELS)
|
||||
def test_voxtral_small_is_an_audio_capable_chat_model(model: str) -> None:
|
||||
main_cost, _ = _load_cost_maps()
|
||||
info = main_cost[model]
|
||||
assert info["mode"] == "chat"
|
||||
assert info["supports_audio_input"] is True
|
||||
assert info["input_cost_per_token"] == pytest.approx(1e-07)
|
||||
assert info["output_cost_per_token"] == pytest.approx(3e-07)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", ALL_VOXTRAL_MODELS)
|
||||
def test_backup_cost_map_matches_main(model: str) -> None:
|
||||
main_cost, backup_cost = _load_cost_maps()
|
||||
assert backup_cost.get(model) == main_cost.get(model)
|
||||
|
||||
|
||||
def test_voxtral_models_are_discoverable_for_the_mistral_wildcard(local_cost_map: None) -> None:
|
||||
"""Regression for #34616: `mistral/*` on the proxy expands to the provider's known models,
|
||||
so voxtral was missing from /v1/models until it landed in the cost map."""
|
||||
provider_models = get_provider_models(provider="mistral") or []
|
||||
assert set(ALL_VOXTRAL_MODELS).issubset(set(provider_models))
|
||||
|
||||
|
||||
def test_voxtral_transcription_cost_is_billed_per_audio_second(local_cost_map: None) -> None:
|
||||
response = TranscriptionResponse(text="demo text")
|
||||
response.duration = 90.0
|
||||
|
||||
cost = litellm.completion_cost(
|
||||
completion_response=response,
|
||||
model="mistral/voxtral-mini-latest",
|
||||
custom_llm_provider="mistral",
|
||||
call_type="atranscription",
|
||||
)
|
||||
|
||||
assert cost == pytest.approx(90.0 * 0.003 / 60)
|
||||
Loading…
Add table
Reference in a new issue