This commit is contained in:
Favour Adesiyan 2026-10-05 10:14:59 +00:00 • committed by GitHub
commit 1acc2c8899
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
4 changed files with 114 additions and 1 deletions

View file

@ -638,6 +638,7 @@ fal_ai_models: Set = set()
fireworks_ai_models: Set = set()
fireworks_ai_embedding_models: Set = set()
deepinfra_models: Set = set()
compactifai_models: Final[Set[str]] = set()
perplexity_models: Set = set()
watsonx_models: Set = set()
gemini_models: Set = set()
@ -840,6 +841,8 @@ def _populate_provider_model_sets(model_cost_map: Dict) -> None:
bedrock_converse_models.add(key)
elif value.get("litellm_provider") == "deepinfra":
deepinfra_models.add(key)
elif value.get("litellm_provider") == "compactifai":
compactifai_models.add(key)
elif value.get("litellm_provider") == "perplexity":
perplexity_models.add(key)
elif value.get("litellm_provider") == "watsonx":
@ -1066,6 +1069,7 @@ model_list = list(
| set(ollama_models)
| bedrock_models
| deepinfra_models
| compactifai_models
| perplexity_models
| set(maritalk_models)
| runwayml_models
@ -1171,6 +1175,7 @@ def _build_models_by_provider() -> dict:
"ollama": ollama_models,
"ollama_chat": ollama_models,
"deepinfra": deepinfra_models,
"compactifai": compactifai_models,
"perplexity": perplexity_models,
"maritalk": maritalk_models,
"watsonx": watsonx_models,

View file

@ -80004,5 +80004,48 @@
"supports_tool_choice": true,
"supports_video_input": false,
"supports_vision": false
},
"compactifai/glm-5-2": {
"input_cost_per_token": 1.1e-06,
"litellm_provider": "compactifai",
"mode": "chat",
"output_cost_per_token": 3.5e-06,
"reasoning_effort_levels": [
"high",
"max"
],
"source": "https://docs.compactif.ai/pricing/",
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true
},
"compactifai/glm-5-3": {
"input_cost_per_token": 1.1e-06,
"litellm_provider": "compactifai",
"mode": "chat",
"output_cost_per_token": 3.5e-06,
"reasoning_effort_levels": [
"low",
"high",
"max"
],
"source": "https://docs.compactif.ai/pricing/",
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true
},
"compactifai/quasar-438b": {
"input_cost_per_token": 6e-07,
"litellm_provider": "compactifai",
"mode": "chat",
"output_cost_per_token": 1.8e-06,
"reasoning_effort_levels": [
"high",
"max"
],
"source": "https://docs.compactif.ai/pricing/",
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true
}
}

View file

@ -80004,5 +80004,48 @@
"supports_tool_choice": true,
"supports_video_input": false,
"supports_vision": false
},
"compactifai/glm-5-2": {
"input_cost_per_token": 1.1e-06,
"litellm_provider": "compactifai",
"mode": "chat",
"output_cost_per_token": 3.5e-06,
"reasoning_effort_levels": [
"high",
"max"
],
"source": "https://docs.compactif.ai/pricing/",
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true
},
"compactifai/glm-5-3": {
"input_cost_per_token": 1.1e-06,
"litellm_provider": "compactifai",
"mode": "chat",
"output_cost_per_token": 3.5e-06,
"reasoning_effort_levels": [
"low",
"high",
"max"
],
"source": "https://docs.compactif.ai/pricing/",
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true
},
"compactifai/quasar-438b": {
"input_cost_per_token": 6e-07,
"litellm_provider": "compactifai",
"mode": "chat",
"output_cost_per_token": 1.8e-06,
"reasoning_effort_levels": [
"high",
"max"
],
"source": "https://docs.compactif.ai/pricing/",
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true
}
}

View file

@ -40,7 +40,7 @@ from litellm.types.utils import (
Usage,
)
from litellm.types.videos.main import VideoObject
from litellm.utils import supports_prompt_caching
from litellm.utils import get_valid_models, supports_prompt_caching
@pytest.fixture
@ -49,6 +49,28 @@ def _local_model_cost_map(monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
@pytest.mark.parametrize("model", ("glm-5-2", "glm-5-3", "quasar-438b"))
def test_compactifai_response_cost_uses_provider_catalog(model: str, _local_model_cost_map: None) -> None:
assert f"compactifai/{model}" in get_valid_models(custom_llm_provider="compactifai")
model_info: Final = litellm.get_model_info(model=model, custom_llm_provider="compactifai")
input_rate: Final = model_info.get("input_cost_per_token")
output_rate: Final = model_info.get("output_cost_per_token")
assert input_rate is not None and input_rate > 0
assert output_rate is not None and output_rate > 0
response: Final = ModelResponse(
model=f"compactifai/{model}",
choices=[],
usage=Usage(prompt_tokens=100, completion_tokens=50, total_tokens=150),
)
assert completion_cost(completion_response=response, custom_llm_provider="compactifai") == pytest.approx(
100 * input_rate + 50 * output_rate
)
assert cost_per_token(
model=model, prompt_tokens=100, completion_tokens=50, custom_llm_provider="compactifai"
) == pytest.approx((100 * input_rate, 50 * output_rate))
def test_cost_per_token_duplicate_openai_prefix_matches_model_cost(monkeypatch):
"""
Router/proxy configs may use deployment ids like openai/openai/<model>. Cost lookup must