mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
Add Tensormesh serverless models to the model cost map (#30037)
* Add Tensormesh serverless models to the model cost map * Flag reasoning support on the Tensormesh models that expose thinking mode
This commit is contained in:
parent
fe60822138
commit
435809aac2
3 changed files with 390 additions and 0 deletions
|
|
@ -41968,5 +41968,164 @@
|
|||
"/v1/audio/transcriptions"
|
||||
],
|
||||
"supports_audio_input": true
|
||||
},
|
||||
"tensormesh/Qwen/Qwen3.5-397B-A17B-FP8": {
|
||||
"litellm_provider": "tensormesh",
|
||||
"mode": "chat",
|
||||
"input_cost_per_token": 6e-07,
|
||||
"output_cost_per_token": 3.6e-06,
|
||||
"cache_read_input_token_cost": 0,
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_reasoning": true,
|
||||
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
|
||||
},
|
||||
"tensormesh/Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8": {
|
||||
"litellm_provider": "tensormesh",
|
||||
"mode": "chat",
|
||||
"input_cost_per_token": 4.5e-07,
|
||||
"output_cost_per_token": 1.8e-06,
|
||||
"cache_read_input_token_cost": 0,
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_system_messages": true,
|
||||
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
|
||||
},
|
||||
"tensormesh/Qwen/Qwen3.6-27B-FP8": {
|
||||
"litellm_provider": "tensormesh",
|
||||
"mode": "chat",
|
||||
"input_cost_per_token": 3.2e-07,
|
||||
"output_cost_per_token": 3.2e-06,
|
||||
"cache_read_input_token_cost": 0,
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_reasoning": true,
|
||||
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
|
||||
},
|
||||
"tensormesh/lukealonso/GLM-5.1-NVFP4-MTP": {
|
||||
"litellm_provider": "tensormesh",
|
||||
"mode": "chat",
|
||||
"input_cost_per_token": 1.4e-06,
|
||||
"output_cost_per_token": 4.4e-06,
|
||||
"cache_read_input_token_cost": 0,
|
||||
"max_input_tokens": 202752,
|
||||
"max_output_tokens": 202752,
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_reasoning": true,
|
||||
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
|
||||
},
|
||||
"tensormesh/deepseek-ai/DeepSeek-V4-Flash": {
|
||||
"litellm_provider": "tensormesh",
|
||||
"mode": "chat",
|
||||
"input_cost_per_token": 1.4e-07,
|
||||
"output_cost_per_token": 2.8e-07,
|
||||
"cache_read_input_token_cost": 0,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_reasoning": true,
|
||||
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
|
||||
},
|
||||
"tensormesh/moonshotai/Kimi-K2.6": {
|
||||
"litellm_provider": "tensormesh",
|
||||
"mode": "chat",
|
||||
"input_cost_per_token": 9.6e-07,
|
||||
"output_cost_per_token": 4e-06,
|
||||
"cache_read_input_token_cost": 0,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_reasoning": true,
|
||||
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
|
||||
},
|
||||
"tensormesh/MiniMaxAI/MiniMax-M2.5": {
|
||||
"litellm_provider": "tensormesh",
|
||||
"mode": "chat",
|
||||
"input_cost_per_token": 3e-07,
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"cache_read_input_token_cost": 0,
|
||||
"max_input_tokens": 196608,
|
||||
"max_output_tokens": 196608,
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_reasoning": true,
|
||||
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
|
||||
},
|
||||
"tensormesh/google/gemma-4-31B-it": {
|
||||
"litellm_provider": "tensormesh",
|
||||
"mode": "chat",
|
||||
"input_cost_per_token": 1.4e-07,
|
||||
"output_cost_per_token": 5.6e-07,
|
||||
"cache_read_input_token_cost": 0,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_reasoning": true,
|
||||
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
|
||||
},
|
||||
"tensormesh/openai/gpt-oss-120b": {
|
||||
"litellm_provider": "tensormesh",
|
||||
"mode": "chat",
|
||||
"input_cost_per_token": 1.5e-07,
|
||||
"output_cost_per_token": 6e-07,
|
||||
"cache_read_input_token_cost": 0,
|
||||
"max_input_tokens": 131072,
|
||||
"max_output_tokens": 131072,
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_reasoning": true,
|
||||
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
|
||||
},
|
||||
"tensormesh/openai/gpt-oss-20b": {
|
||||
"litellm_provider": "tensormesh",
|
||||
"mode": "chat",
|
||||
"input_cost_per_token": 7e-08,
|
||||
"output_cost_per_token": 2.8e-07,
|
||||
"cache_read_input_token_cost": 0,
|
||||
"max_input_tokens": 131072,
|
||||
"max_output_tokens": 131072,
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_reasoning": true,
|
||||
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
|
||||
}
|
||||
}
|
||||
|
|
@ -42180,5 +42180,164 @@
|
|||
"source": "https://soniox.com/pricing",
|
||||
"supported_endpoints": ["/v1/audio/transcriptions"],
|
||||
"supports_audio_input": true
|
||||
},
|
||||
"tensormesh/Qwen/Qwen3.5-397B-A17B-FP8": {
|
||||
"litellm_provider": "tensormesh",
|
||||
"mode": "chat",
|
||||
"input_cost_per_token": 6e-07,
|
||||
"output_cost_per_token": 3.6e-06,
|
||||
"cache_read_input_token_cost": 0,
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_reasoning": true,
|
||||
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
|
||||
},
|
||||
"tensormesh/Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8": {
|
||||
"litellm_provider": "tensormesh",
|
||||
"mode": "chat",
|
||||
"input_cost_per_token": 4.5e-07,
|
||||
"output_cost_per_token": 1.8e-06,
|
||||
"cache_read_input_token_cost": 0,
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_system_messages": true,
|
||||
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
|
||||
},
|
||||
"tensormesh/Qwen/Qwen3.6-27B-FP8": {
|
||||
"litellm_provider": "tensormesh",
|
||||
"mode": "chat",
|
||||
"input_cost_per_token": 3.2e-07,
|
||||
"output_cost_per_token": 3.2e-06,
|
||||
"cache_read_input_token_cost": 0,
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_reasoning": true,
|
||||
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
|
||||
},
|
||||
"tensormesh/lukealonso/GLM-5.1-NVFP4-MTP": {
|
||||
"litellm_provider": "tensormesh",
|
||||
"mode": "chat",
|
||||
"input_cost_per_token": 1.4e-06,
|
||||
"output_cost_per_token": 4.4e-06,
|
||||
"cache_read_input_token_cost": 0,
|
||||
"max_input_tokens": 202752,
|
||||
"max_output_tokens": 202752,
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_reasoning": true,
|
||||
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
|
||||
},
|
||||
"tensormesh/deepseek-ai/DeepSeek-V4-Flash": {
|
||||
"litellm_provider": "tensormesh",
|
||||
"mode": "chat",
|
||||
"input_cost_per_token": 1.4e-07,
|
||||
"output_cost_per_token": 2.8e-07,
|
||||
"cache_read_input_token_cost": 0,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_reasoning": true,
|
||||
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
|
||||
},
|
||||
"tensormesh/moonshotai/Kimi-K2.6": {
|
||||
"litellm_provider": "tensormesh",
|
||||
"mode": "chat",
|
||||
"input_cost_per_token": 9.6e-07,
|
||||
"output_cost_per_token": 4e-06,
|
||||
"cache_read_input_token_cost": 0,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_reasoning": true,
|
||||
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
|
||||
},
|
||||
"tensormesh/MiniMaxAI/MiniMax-M2.5": {
|
||||
"litellm_provider": "tensormesh",
|
||||
"mode": "chat",
|
||||
"input_cost_per_token": 3e-07,
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"cache_read_input_token_cost": 0,
|
||||
"max_input_tokens": 196608,
|
||||
"max_output_tokens": 196608,
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_reasoning": true,
|
||||
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
|
||||
},
|
||||
"tensormesh/google/gemma-4-31B-it": {
|
||||
"litellm_provider": "tensormesh",
|
||||
"mode": "chat",
|
||||
"input_cost_per_token": 1.4e-07,
|
||||
"output_cost_per_token": 5.6e-07,
|
||||
"cache_read_input_token_cost": 0,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_reasoning": true,
|
||||
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
|
||||
},
|
||||
"tensormesh/openai/gpt-oss-120b": {
|
||||
"litellm_provider": "tensormesh",
|
||||
"mode": "chat",
|
||||
"input_cost_per_token": 1.5e-07,
|
||||
"output_cost_per_token": 6e-07,
|
||||
"cache_read_input_token_cost": 0,
|
||||
"max_input_tokens": 131072,
|
||||
"max_output_tokens": 131072,
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_reasoning": true,
|
||||
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
|
||||
},
|
||||
"tensormesh/openai/gpt-oss-20b": {
|
||||
"litellm_provider": "tensormesh",
|
||||
"mode": "chat",
|
||||
"input_cost_per_token": 7e-08,
|
||||
"output_cost_per_token": 2.8e-07,
|
||||
"cache_read_input_token_cost": 0,
|
||||
"max_input_tokens": 131072,
|
||||
"max_output_tokens": 131072,
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_reasoning": true,
|
||||
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -2,8 +2,23 @@
|
|||
Tests for Tensormesh provider configuration and integration.
|
||||
"""
|
||||
|
||||
import pytest
|
||||
|
||||
import litellm
|
||||
|
||||
TENSORMESH_MODELS = [
|
||||
"tensormesh/Qwen/Qwen3.5-397B-A17B-FP8",
|
||||
"tensormesh/Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8",
|
||||
"tensormesh/Qwen/Qwen3.6-27B-FP8",
|
||||
"tensormesh/lukealonso/GLM-5.1-NVFP4-MTP",
|
||||
"tensormesh/deepseek-ai/DeepSeek-V4-Flash",
|
||||
"tensormesh/moonshotai/Kimi-K2.6",
|
||||
"tensormesh/MiniMaxAI/MiniMax-M2.5",
|
||||
"tensormesh/google/gemma-4-31B-it",
|
||||
"tensormesh/openai/gpt-oss-120b",
|
||||
"tensormesh/openai/gpt-oss-20b",
|
||||
]
|
||||
|
||||
|
||||
class TestTensormeshProviderConfig:
|
||||
"""Test Tensormesh provider configuration"""
|
||||
|
|
@ -82,3 +97,60 @@ class TestTensormeshProviderConfig:
|
|||
|
||||
assert len(router.model_list) == 1
|
||||
assert router.model_list[0]["model_name"] == "tensormesh-chat"
|
||||
|
||||
|
||||
class TestTensormeshCostMap:
|
||||
"""The serverless models are registered in the cost map so LiteLLM can
|
||||
price requests and unblock tool-calling params on the JSON provider path."""
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _use_local_model_cost_map(self, monkeypatch):
|
||||
original_model_cost = litellm.model_cost
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
litellm.model_cost = litellm.get_model_cost_map(url="")
|
||||
litellm.get_model_info.cache_clear()
|
||||
try:
|
||||
yield
|
||||
finally:
|
||||
litellm.model_cost = original_model_cost
|
||||
litellm.get_model_info.cache_clear()
|
||||
|
||||
def test_models_registered_with_capabilities(self):
|
||||
for model in TENSORMESH_MODELS:
|
||||
info = litellm.get_model_info(model)
|
||||
assert info["litellm_provider"] == "tensormesh"
|
||||
assert info["mode"] == "chat"
|
||||
assert litellm.supports_function_calling(model) is True, model
|
||||
assert litellm.supports_response_schema(model) is True, model
|
||||
assert litellm.model_cost[model]["supports_tool_choice"] is True, model
|
||||
assert litellm.model_cost[model]["supports_prompt_caching"] is True, model
|
||||
|
||||
def test_reasoning_flag_matches_expected_set(self):
|
||||
reasoning_models = {
|
||||
"tensormesh/deepseek-ai/DeepSeek-V4-Flash",
|
||||
"tensormesh/Qwen/Qwen3.5-397B-A17B-FP8",
|
||||
"tensormesh/Qwen/Qwen3.6-27B-FP8",
|
||||
"tensormesh/lukealonso/GLM-5.1-NVFP4-MTP",
|
||||
"tensormesh/MiniMaxAI/MiniMax-M2.5",
|
||||
"tensormesh/moonshotai/Kimi-K2.6",
|
||||
"tensormesh/openai/gpt-oss-120b",
|
||||
"tensormesh/openai/gpt-oss-20b",
|
||||
"tensormesh/google/gemma-4-31B-it",
|
||||
}
|
||||
for model in TENSORMESH_MODELS:
|
||||
assert litellm.supports_reasoning(model) is (model in reasoning_models), model
|
||||
|
||||
def test_cost_is_wired_and_cache_reads_are_free(self):
|
||||
prompt_cost, completion_cost = litellm.cost_per_token(
|
||||
model="tensormesh/openai/gpt-oss-120b",
|
||||
prompt_tokens=1_000_000,
|
||||
completion_tokens=1_000_000,
|
||||
)
|
||||
assert prompt_cost == pytest.approx(0.15)
|
||||
assert completion_cost == pytest.approx(0.60)
|
||||
assert (
|
||||
litellm.model_cost["tensormesh/openai/gpt-oss-120b"][
|
||||
"cache_read_input_token_cost"
|
||||
]
|
||||
== 0
|
||||
)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue