Add Tensormesh serverless models to the model cost map (#30037)

* Add Tensormesh serverless models to the model cost map

* Flag reasoning support on the Tensormesh models that expose thinking mode
This commit is contained in:
daitran-tensormesh 2026-06-10 17:06:10 +07:00 • committed by Sameer Kankute
parent fe60822138
commit 435809aac2
No known key found for this signature in database
3 changed files with 390 additions and 0 deletions

View file

@ -41968,5 +41968,164 @@
"/v1/audio/transcriptions"
],
"supports_audio_input": true
},
"tensormesh/Qwen/Qwen3.5-397B-A17B-FP8": {
"litellm_provider": "tensormesh",
"mode": "chat",
"input_cost_per_token": 6e-07,
"output_cost_per_token": 3.6e-06,
"cache_read_input_token_cost": 0,
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_response_schema": true,
"supports_prompt_caching": true,
"supports_system_messages": true,
"supports_reasoning": true,
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
},
"tensormesh/Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8": {
"litellm_provider": "tensormesh",
"mode": "chat",
"input_cost_per_token": 4.5e-07,
"output_cost_per_token": 1.8e-06,
"cache_read_input_token_cost": 0,
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_response_schema": true,
"supports_prompt_caching": true,
"supports_system_messages": true,
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
},
"tensormesh/Qwen/Qwen3.6-27B-FP8": {
"litellm_provider": "tensormesh",
"mode": "chat",
"input_cost_per_token": 3.2e-07,
"output_cost_per_token": 3.2e-06,
"cache_read_input_token_cost": 0,
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_response_schema": true,
"supports_prompt_caching": true,
"supports_system_messages": true,
"supports_reasoning": true,
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
},
"tensormesh/lukealonso/GLM-5.1-NVFP4-MTP": {
"litellm_provider": "tensormesh",
"mode": "chat",
"input_cost_per_token": 1.4e-06,
"output_cost_per_token": 4.4e-06,
"cache_read_input_token_cost": 0,
"max_input_tokens": 202752,
"max_output_tokens": 202752,
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_response_schema": true,
"supports_prompt_caching": true,
"supports_system_messages": true,
"supports_reasoning": true,
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
},
"tensormesh/deepseek-ai/DeepSeek-V4-Flash": {
"litellm_provider": "tensormesh",
"mode": "chat",
"input_cost_per_token": 1.4e-07,
"output_cost_per_token": 2.8e-07,
"cache_read_input_token_cost": 0,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_response_schema": true,
"supports_prompt_caching": true,
"supports_system_messages": true,
"supports_reasoning": true,
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
},
"tensormesh/moonshotai/Kimi-K2.6": {
"litellm_provider": "tensormesh",
"mode": "chat",
"input_cost_per_token": 9.6e-07,
"output_cost_per_token": 4e-06,
"cache_read_input_token_cost": 0,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_response_schema": true,
"supports_prompt_caching": true,
"supports_system_messages": true,
"supports_reasoning": true,
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
},
"tensormesh/MiniMaxAI/MiniMax-M2.5": {
"litellm_provider": "tensormesh",
"mode": "chat",
"input_cost_per_token": 3e-07,
"output_cost_per_token": 1.2e-06,
"cache_read_input_token_cost": 0,
"max_input_tokens": 196608,
"max_output_tokens": 196608,
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_response_schema": true,
"supports_prompt_caching": true,
"supports_system_messages": true,
"supports_reasoning": true,
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
},
"tensormesh/google/gemma-4-31B-it": {
"litellm_provider": "tensormesh",
"mode": "chat",
"input_cost_per_token": 1.4e-07,
"output_cost_per_token": 5.6e-07,
"cache_read_input_token_cost": 0,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_response_schema": true,
"supports_prompt_caching": true,
"supports_system_messages": true,
"supports_reasoning": true,
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
},
"tensormesh/openai/gpt-oss-120b": {
"litellm_provider": "tensormesh",
"mode": "chat",
"input_cost_per_token": 1.5e-07,
"output_cost_per_token": 6e-07,
"cache_read_input_token_cost": 0,
"max_input_tokens": 131072,
"max_output_tokens": 131072,
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_response_schema": true,
"supports_prompt_caching": true,
"supports_system_messages": true,
"supports_reasoning": true,
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
},
"tensormesh/openai/gpt-oss-20b": {
"litellm_provider": "tensormesh",
"mode": "chat",
"input_cost_per_token": 7e-08,
"output_cost_per_token": 2.8e-07,
"cache_read_input_token_cost": 0,
"max_input_tokens": 131072,
"max_output_tokens": 131072,
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_response_schema": true,
"supports_prompt_caching": true,
"supports_system_messages": true,
"supports_reasoning": true,
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
}
}

View file

@ -42180,5 +42180,164 @@
"source": "https://soniox.com/pricing",
"supported_endpoints": ["/v1/audio/transcriptions"],
"supports_audio_input": true
},
"tensormesh/Qwen/Qwen3.5-397B-A17B-FP8": {
"litellm_provider": "tensormesh",
"mode": "chat",
"input_cost_per_token": 6e-07,
"output_cost_per_token": 3.6e-06,
"cache_read_input_token_cost": 0,
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_response_schema": true,
"supports_prompt_caching": true,
"supports_system_messages": true,
"supports_reasoning": true,
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
},
"tensormesh/Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8": {
"litellm_provider": "tensormesh",
"mode": "chat",
"input_cost_per_token": 4.5e-07,
"output_cost_per_token": 1.8e-06,
"cache_read_input_token_cost": 0,
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_response_schema": true,
"supports_prompt_caching": true,
"supports_system_messages": true,
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
},
"tensormesh/Qwen/Qwen3.6-27B-FP8": {
"litellm_provider": "tensormesh",
"mode": "chat",
"input_cost_per_token": 3.2e-07,
"output_cost_per_token": 3.2e-06,
"cache_read_input_token_cost": 0,
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_response_schema": true,
"supports_prompt_caching": true,
"supports_system_messages": true,
"supports_reasoning": true,
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
},
"tensormesh/lukealonso/GLM-5.1-NVFP4-MTP": {
"litellm_provider": "tensormesh",
"mode": "chat",
"input_cost_per_token": 1.4e-06,
"output_cost_per_token": 4.4e-06,
"cache_read_input_token_cost": 0,
"max_input_tokens": 202752,
"max_output_tokens": 202752,
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_response_schema": true,
"supports_prompt_caching": true,
"supports_system_messages": true,
"supports_reasoning": true,
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
},
"tensormesh/deepseek-ai/DeepSeek-V4-Flash": {
"litellm_provider": "tensormesh",
"mode": "chat",
"input_cost_per_token": 1.4e-07,
"output_cost_per_token": 2.8e-07,
"cache_read_input_token_cost": 0,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_response_schema": true,
"supports_prompt_caching": true,
"supports_system_messages": true,
"supports_reasoning": true,
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
},
"tensormesh/moonshotai/Kimi-K2.6": {
"litellm_provider": "tensormesh",
"mode": "chat",
"input_cost_per_token": 9.6e-07,
"output_cost_per_token": 4e-06,
"cache_read_input_token_cost": 0,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_response_schema": true,
"supports_prompt_caching": true,
"supports_system_messages": true,
"supports_reasoning": true,
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
},
"tensormesh/MiniMaxAI/MiniMax-M2.5": {
"litellm_provider": "tensormesh",
"mode": "chat",
"input_cost_per_token": 3e-07,
"output_cost_per_token": 1.2e-06,
"cache_read_input_token_cost": 0,
"max_input_tokens": 196608,
"max_output_tokens": 196608,
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_response_schema": true,
"supports_prompt_caching": true,
"supports_system_messages": true,
"supports_reasoning": true,
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
},
"tensormesh/google/gemma-4-31B-it": {
"litellm_provider": "tensormesh",
"mode": "chat",
"input_cost_per_token": 1.4e-07,
"output_cost_per_token": 5.6e-07,
"cache_read_input_token_cost": 0,
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_response_schema": true,
"supports_prompt_caching": true,
"supports_system_messages": true,
"supports_reasoning": true,
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
},
"tensormesh/openai/gpt-oss-120b": {
"litellm_provider": "tensormesh",
"mode": "chat",
"input_cost_per_token": 1.5e-07,
"output_cost_per_token": 6e-07,
"cache_read_input_token_cost": 0,
"max_input_tokens": 131072,
"max_output_tokens": 131072,
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_response_schema": true,
"supports_prompt_caching": true,
"supports_system_messages": true,
"supports_reasoning": true,
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
},
"tensormesh/openai/gpt-oss-20b": {
"litellm_provider": "tensormesh",
"mode": "chat",
"input_cost_per_token": 7e-08,
"output_cost_per_token": 2.8e-07,
"cache_read_input_token_cost": 0,
"max_input_tokens": 131072,
"max_output_tokens": 131072,
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_response_schema": true,
"supports_prompt_caching": true,
"supports_system_messages": true,
"supports_reasoning": true,
"source": "https://serverless.tensormesh.ai/v1/models/openrouter"
}
}

View file

@ -2,8 +2,23 @@
Tests for Tensormesh provider configuration and integration.
"""
import pytest
import litellm
TENSORMESH_MODELS = [
"tensormesh/Qwen/Qwen3.5-397B-A17B-FP8",
"tensormesh/Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8",
"tensormesh/Qwen/Qwen3.6-27B-FP8",
"tensormesh/lukealonso/GLM-5.1-NVFP4-MTP",
"tensormesh/deepseek-ai/DeepSeek-V4-Flash",
"tensormesh/moonshotai/Kimi-K2.6",
"tensormesh/MiniMaxAI/MiniMax-M2.5",
"tensormesh/google/gemma-4-31B-it",
"tensormesh/openai/gpt-oss-120b",
"tensormesh/openai/gpt-oss-20b",
]
class TestTensormeshProviderConfig:
"""Test Tensormesh provider configuration"""
@ -82,3 +97,60 @@ class TestTensormeshProviderConfig:
assert len(router.model_list) == 1
assert router.model_list[0]["model_name"] == "tensormesh-chat"
class TestTensormeshCostMap:
"""The serverless models are registered in the cost map so LiteLLM can
price requests and unblock tool-calling params on the JSON provider path."""
@pytest.fixture(autouse=True)
def _use_local_model_cost_map(self, monkeypatch):
original_model_cost = litellm.model_cost
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
litellm.model_cost = litellm.get_model_cost_map(url="")
litellm.get_model_info.cache_clear()
try:
yield
finally:
litellm.model_cost = original_model_cost
litellm.get_model_info.cache_clear()
def test_models_registered_with_capabilities(self):
for model in TENSORMESH_MODELS:
info = litellm.get_model_info(model)
assert info["litellm_provider"] == "tensormesh"
assert info["mode"] == "chat"
assert litellm.supports_function_calling(model) is True, model
assert litellm.supports_response_schema(model) is True, model
assert litellm.model_cost[model]["supports_tool_choice"] is True, model
assert litellm.model_cost[model]["supports_prompt_caching"] is True, model
def test_reasoning_flag_matches_expected_set(self):
reasoning_models = {
"tensormesh/deepseek-ai/DeepSeek-V4-Flash",
"tensormesh/Qwen/Qwen3.5-397B-A17B-FP8",
"tensormesh/Qwen/Qwen3.6-27B-FP8",
"tensormesh/lukealonso/GLM-5.1-NVFP4-MTP",
"tensormesh/MiniMaxAI/MiniMax-M2.5",
"tensormesh/moonshotai/Kimi-K2.6",
"tensormesh/openai/gpt-oss-120b",
"tensormesh/openai/gpt-oss-20b",
"tensormesh/google/gemma-4-31B-it",
}
for model in TENSORMESH_MODELS:
assert litellm.supports_reasoning(model) is (model in reasoning_models), model
def test_cost_is_wired_and_cache_reads_are_free(self):
prompt_cost, completion_cost = litellm.cost_per_token(
model="tensormesh/openai/gpt-oss-120b",
prompt_tokens=1_000_000,
completion_tokens=1_000_000,
)
assert prompt_cost == pytest.approx(0.15)
assert completion_cost == pytest.approx(0.60)
assert (
litellm.model_cost["tensormesh/openai/gpt-oss-120b"][
"cache_read_input_token_cost"
]
== 0
)