From 5d895c6facf6e9a7deeae69359a011523c4c5763 Mon Sep 17 00:00:00 2001 From: Favour Adesiyan <101547081+favouradesiyan@users.noreply.github.com> Date: Sun, 27 Sep 2026 20:41:18 +0200 Subject: [PATCH] feat: add CompactifAI model pricing and capabilities --- litellm/__init__.py | 5 +++ ...odel_prices_and_context_window_backup.json | 43 +++++++++++++++++++ model_prices_and_context_window.json | 43 +++++++++++++++++++ tests/unit/test_cost_calculator.py | 24 ++++++++++- 4 files changed, 114 insertions(+), 1 deletion(-) diff --git a/litellm/__init__.py b/litellm/__init__.py index 5d10737e876..64254e64ba1 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -636,6 +636,7 @@ fal_ai_models: Set = set() fireworks_ai_models: Set = set() fireworks_ai_embedding_models: Set = set() deepinfra_models: Set = set() +compactifai_models: Final[Set[str]] = set() # mutable-ok: provider registry is populated and refreshed in place perplexity_models: Set = set() watsonx_models: Set = set() gemini_models: Set = set() @@ -838,6 +839,8 @@ def _populate_provider_model_sets(model_cost_map: Dict) -> None: bedrock_converse_models.add(key) elif value.get("litellm_provider") == "deepinfra": deepinfra_models.add(key) + elif value.get("litellm_provider") == "compactifai": + compactifai_models.add(key) elif value.get("litellm_provider") == "perplexity": perplexity_models.add(key) elif value.get("litellm_provider") == "watsonx": @@ -1064,6 +1067,7 @@ model_list = list( | set(ollama_models) | bedrock_models | deepinfra_models + | compactifai_models | perplexity_models | set(maritalk_models) | runwayml_models @@ -1169,6 +1173,7 @@ def _build_models_by_provider() -> dict: "ollama": ollama_models, "ollama_chat": ollama_models, "deepinfra": deepinfra_models, + "compactifai": compactifai_models, "perplexity": perplexity_models, "maritalk": maritalk_models, "watsonx": watsonx_models, diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index b36d84ea027..7fc999d8be5 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -77910,5 +77910,48 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_tool_choice": true + }, + "compactifai/glm-5-2": { + "input_cost_per_token": 1.1e-06, + "litellm_provider": "compactifai", + "mode": "chat", + "output_cost_per_token": 3.5e-06, + "reasoning_effort_levels": [ + "high", + "max" + ], + "source": "https://docs.compactif.ai/pricing/", + "supports_function_calling": true, + "supports_reasoning": true, + "supports_response_schema": true + }, + "compactifai/glm-5-3": { + "input_cost_per_token": 1.1e-06, + "litellm_provider": "compactifai", + "mode": "chat", + "output_cost_per_token": 3.5e-06, + "reasoning_effort_levels": [ + "low", + "high", + "max" + ], + "source": "https://docs.compactif.ai/pricing/", + "supports_function_calling": true, + "supports_reasoning": true, + "supports_response_schema": true + }, + "compactifai/quasar-438b": { + "input_cost_per_token": 6e-07, + "litellm_provider": "compactifai", + "mode": "chat", + "output_cost_per_token": 1.8e-06, + "reasoning_effort_levels": [ + "high", + "max" + ], + "source": "https://docs.compactif.ai/pricing/", + "supports_function_calling": true, + "supports_reasoning": true, + "supports_response_schema": true } } diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index b36d84ea027..7fc999d8be5 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -77910,5 +77910,48 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_tool_choice": true + }, + "compactifai/glm-5-2": { + "input_cost_per_token": 1.1e-06, + "litellm_provider": "compactifai", + "mode": "chat", + "output_cost_per_token": 3.5e-06, + "reasoning_effort_levels": [ + "high", + "max" + ], + "source": "https://docs.compactif.ai/pricing/", + "supports_function_calling": true, + "supports_reasoning": true, + "supports_response_schema": true + }, + "compactifai/glm-5-3": { + "input_cost_per_token": 1.1e-06, + "litellm_provider": "compactifai", + "mode": "chat", + "output_cost_per_token": 3.5e-06, + "reasoning_effort_levels": [ + "low", + "high", + "max" + ], + "source": "https://docs.compactif.ai/pricing/", + "supports_function_calling": true, + "supports_reasoning": true, + "supports_response_schema": true + }, + "compactifai/quasar-438b": { + "input_cost_per_token": 6e-07, + "litellm_provider": "compactifai", + "mode": "chat", + "output_cost_per_token": 1.8e-06, + "reasoning_effort_levels": [ + "high", + "max" + ], + "source": "https://docs.compactif.ai/pricing/", + "supports_function_calling": true, + "supports_reasoning": true, + "supports_response_schema": true } } diff --git a/tests/unit/test_cost_calculator.py b/tests/unit/test_cost_calculator.py index 62ef9f11c2e..07d52ddf832 100644 --- a/tests/unit/test_cost_calculator.py +++ b/tests/unit/test_cost_calculator.py @@ -38,7 +38,7 @@ from litellm.types.utils import ( Usage, ) from litellm.types.videos.main import VideoObject -from litellm.utils import supports_prompt_caching +from litellm.utils import get_valid_models, supports_prompt_caching @pytest.fixture @@ -47,6 +47,28 @@ def _local_model_cost_map(monkeypatch: pytest.MonkeyPatch) -> None: monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) +@pytest.mark.parametrize("model", ("glm-5-2", "glm-5-3", "quasar-438b")) +def test_compactifai_response_cost_uses_provider_catalog(model: str, _local_model_cost_map: None) -> None: + assert f"compactifai/{model}" in get_valid_models(custom_llm_provider="compactifai") + model_info: Final = litellm.get_model_info(model=model, custom_llm_provider="compactifai") + input_rate: Final = model_info.get("input_cost_per_token") + output_rate: Final = model_info.get("output_cost_per_token") + assert input_rate is not None and input_rate > 0 + assert output_rate is not None and output_rate > 0 + + response: Final = ModelResponse( + model=f"compactifai/{model}", + choices=[], + usage=Usage(prompt_tokens=100, completion_tokens=50, total_tokens=150), + ) + assert completion_cost(completion_response=response, custom_llm_provider="compactifai") == pytest.approx( + 100 * input_rate + 50 * output_rate + ) + assert cost_per_token( + model=model, prompt_tokens=100, completion_tokens=50, custom_llm_provider="compactifai" + ) == pytest.approx((100 * input_rate, 50 * output_rate)) + + def test_cost_per_token_duplicate_openai_prefix_matches_model_cost(monkeypatch): """ Router/proxy configs may use deployment ids like openai/openai/. Cost lookup must