mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-11 03:38:38 +00:00
feat: add CompactifAI model pricing and capabilities
This commit is contained in:
parent
22b36cbcf6
commit
5d895c6fac
4 changed files with 114 additions and 1 deletions
|
|
@ -636,6 +636,7 @@ fal_ai_models: Set = set()
|
|||
fireworks_ai_models: Set = set()
|
||||
fireworks_ai_embedding_models: Set = set()
|
||||
deepinfra_models: Set = set()
|
||||
compactifai_models: Final[Set[str]] = set() # mutable-ok: provider registry is populated and refreshed in place
|
||||
perplexity_models: Set = set()
|
||||
watsonx_models: Set = set()
|
||||
gemini_models: Set = set()
|
||||
|
|
@ -838,6 +839,8 @@ def _populate_provider_model_sets(model_cost_map: Dict) -> None:
|
|||
bedrock_converse_models.add(key)
|
||||
elif value.get("litellm_provider") == "deepinfra":
|
||||
deepinfra_models.add(key)
|
||||
elif value.get("litellm_provider") == "compactifai":
|
||||
compactifai_models.add(key)
|
||||
elif value.get("litellm_provider") == "perplexity":
|
||||
perplexity_models.add(key)
|
||||
elif value.get("litellm_provider") == "watsonx":
|
||||
|
|
@ -1064,6 +1067,7 @@ model_list = list(
|
|||
| set(ollama_models)
|
||||
| bedrock_models
|
||||
| deepinfra_models
|
||||
| compactifai_models
|
||||
| perplexity_models
|
||||
| set(maritalk_models)
|
||||
| runwayml_models
|
||||
|
|
@ -1169,6 +1173,7 @@ def _build_models_by_provider() -> dict:
|
|||
"ollama": ollama_models,
|
||||
"ollama_chat": ollama_models,
|
||||
"deepinfra": deepinfra_models,
|
||||
"compactifai": compactifai_models,
|
||||
"perplexity": perplexity_models,
|
||||
"maritalk": maritalk_models,
|
||||
"watsonx": watsonx_models,
|
||||
|
|
|
|||
|
|
@ -77910,5 +77910,48 @@
|
|||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"compactifai/glm-5-2": {
|
||||
"input_cost_per_token": 1.1e-06,
|
||||
"litellm_provider": "compactifai",
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3.5e-06,
|
||||
"reasoning_effort_levels": [
|
||||
"high",
|
||||
"max"
|
||||
],
|
||||
"source": "https://docs.compactif.ai/pricing/",
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true
|
||||
},
|
||||
"compactifai/glm-5-3": {
|
||||
"input_cost_per_token": 1.1e-06,
|
||||
"litellm_provider": "compactifai",
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3.5e-06,
|
||||
"reasoning_effort_levels": [
|
||||
"low",
|
||||
"high",
|
||||
"max"
|
||||
],
|
||||
"source": "https://docs.compactif.ai/pricing/",
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true
|
||||
},
|
||||
"compactifai/quasar-438b": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "compactifai",
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.8e-06,
|
||||
"reasoning_effort_levels": [
|
||||
"high",
|
||||
"max"
|
||||
],
|
||||
"source": "https://docs.compactif.ai/pricing/",
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -77910,5 +77910,48 @@
|
|||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"compactifai/glm-5-2": {
|
||||
"input_cost_per_token": 1.1e-06,
|
||||
"litellm_provider": "compactifai",
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3.5e-06,
|
||||
"reasoning_effort_levels": [
|
||||
"high",
|
||||
"max"
|
||||
],
|
||||
"source": "https://docs.compactif.ai/pricing/",
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true
|
||||
},
|
||||
"compactifai/glm-5-3": {
|
||||
"input_cost_per_token": 1.1e-06,
|
||||
"litellm_provider": "compactifai",
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3.5e-06,
|
||||
"reasoning_effort_levels": [
|
||||
"low",
|
||||
"high",
|
||||
"max"
|
||||
],
|
||||
"source": "https://docs.compactif.ai/pricing/",
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true
|
||||
},
|
||||
"compactifai/quasar-438b": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "compactifai",
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.8e-06,
|
||||
"reasoning_effort_levels": [
|
||||
"high",
|
||||
"max"
|
||||
],
|
||||
"source": "https://docs.compactif.ai/pricing/",
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -38,7 +38,7 @@ from litellm.types.utils import (
|
|||
Usage,
|
||||
)
|
||||
from litellm.types.videos.main import VideoObject
|
||||
from litellm.utils import supports_prompt_caching
|
||||
from litellm.utils import get_valid_models, supports_prompt_caching
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
|
|
@ -47,6 +47,28 @@ def _local_model_cost_map(monkeypatch: pytest.MonkeyPatch) -> None:
|
|||
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", ("glm-5-2", "glm-5-3", "quasar-438b"))
|
||||
def test_compactifai_response_cost_uses_provider_catalog(model: str, _local_model_cost_map: None) -> None:
|
||||
assert f"compactifai/{model}" in get_valid_models(custom_llm_provider="compactifai")
|
||||
model_info: Final = litellm.get_model_info(model=model, custom_llm_provider="compactifai")
|
||||
input_rate: Final = model_info.get("input_cost_per_token")
|
||||
output_rate: Final = model_info.get("output_cost_per_token")
|
||||
assert input_rate is not None and input_rate > 0
|
||||
assert output_rate is not None and output_rate > 0
|
||||
|
||||
response: Final = ModelResponse(
|
||||
model=f"compactifai/{model}",
|
||||
choices=[],
|
||||
usage=Usage(prompt_tokens=100, completion_tokens=50, total_tokens=150),
|
||||
)
|
||||
assert completion_cost(completion_response=response, custom_llm_provider="compactifai") == pytest.approx(
|
||||
100 * input_rate + 50 * output_rate
|
||||
)
|
||||
assert cost_per_token(
|
||||
model=model, prompt_tokens=100, completion_tokens=50, custom_llm_provider="compactifai"
|
||||
) == pytest.approx((100 * input_rate, 50 * output_rate))
|
||||
|
||||
|
||||
def test_cost_per_token_duplicate_openai_prefix_matches_model_cost(monkeypatch):
|
||||
"""
|
||||
Router/proxy configs may use deployment ids like openai/openai/<model>. Cost lookup must
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue