From bf1b1c934b69469f64a01ca0d4a066953cd825cc Mon Sep 17 00:00:00 2001 From: rajitkhanna Date: Fri, 11 Sep 2026 13:15:13 -0700 Subject: [PATCH] feat(providers): add Prism provider --- litellm/constants.py | 2 + litellm/llms/openai_like/providers.json | 6 ++ ...odel_prices_and_context_window_backup.json | 22 +++++++ litellm/types/utils.py | 1 + model_prices_and_context_window.json | 22 +++++++ .../llms/openai_like/test_prism_provider.py | 60 +++++++++++++++++++ 6 files changed, 113 insertions(+) create mode 100644 tests/test_litellm/llms/openai_like/test_prism_provider.py diff --git a/litellm/constants.py b/litellm/constants.py index 6b984c2673c..11a9574b80a 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -894,6 +894,7 @@ openai_compatible_endpoints: Final[list] = [ "https://api.meta.ai/v1", "https://api.cognition.ai/v1", "https://api.scx.ai/v1", + "https://api.prisminference.com/v1", "https://gigachat.devices.sberbank.ru/api/v1", ] @@ -966,6 +967,7 @@ openai_compatible_providers: Final[list] = [ "meta", # Meta Model API (Muse Spark) - JSON-configured provider "cognition", "scx-ai", + "prism", ] openai_text_completion_compatible_providers: Final[list] = [ # providers that support `/v1/completions` "together_ai", diff --git a/litellm/llms/openai_like/providers.json b/litellm/llms/openai_like/providers.json index a458a209ea9..ddf63ae5829 100644 --- a/litellm/llms/openai_like/providers.json +++ b/litellm/llms/openai_like/providers.json @@ -200,5 +200,11 @@ "temperature_max": 1.99 }, "supported_endpoints": ["/v1/chat/completions"] + }, + "prism": { + "base_url": "https://api.prisminference.com/v1", + "api_key_env": "PRISM_API_KEY", + "api_base_env": "PRISM_API_BASE", + "supported_endpoints": ["/v1/chat/completions"] } } diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index ad992ff92eb..b9efc620a1d 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -64677,5 +64677,27 @@ "supports_tool_choice": false, "supports_response_schema": true, "supports_vision": false + }, + "prism/deepseek-v4-flash": { + "cache_read_input_token_cost": 7e-08, + "input_cost_per_token": 1.4e-07, + "litellm_provider": "prism", + "max_input_tokens": 1000000, + "max_output_tokens": 393216, + "max_tokens": 393216, + "mode": "chat", + "output_cost_per_token": 2.8e-07, + "source": "https://prisminference.com/pricing", + "supported_endpoints": [ + "/v1/chat/completions" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true } } diff --git a/litellm/types/utils.py b/litellm/types/utils.py index e3ea37dc0c8..d90233613e0 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -4012,6 +4012,7 @@ class LlmProviders(str, Enum): PINSTRIPES = "pinstripes" COGNITION = "cognition" SCX_AI = "scx-ai" + PRISM = "prism" DARKBLOOM = "darkbloom" META = "meta" LITELLM_AGENT = "litellm_agent" diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index ad992ff92eb..b9efc620a1d 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -64677,5 +64677,27 @@ "supports_tool_choice": false, "supports_response_schema": true, "supports_vision": false + }, + "prism/deepseek-v4-flash": { + "cache_read_input_token_cost": 7e-08, + "input_cost_per_token": 1.4e-07, + "litellm_provider": "prism", + "max_input_tokens": 1000000, + "max_output_tokens": 393216, + "max_tokens": 393216, + "mode": "chat", + "output_cost_per_token": 2.8e-07, + "source": "https://prisminference.com/pricing", + "supported_endpoints": [ + "/v1/chat/completions" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true } } diff --git a/tests/test_litellm/llms/openai_like/test_prism_provider.py b/tests/test_litellm/llms/openai_like/test_prism_provider.py new file mode 100644 index 00000000000..c308bd49aa1 --- /dev/null +++ b/tests/test_litellm/llms/openai_like/test_prism_provider.py @@ -0,0 +1,60 @@ +import pytest + +import litellm + + +def test_prism_provider_resolution(monkeypatch: pytest.MonkeyPatch): + from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider + + monkeypatch.setenv("PRISM_API_KEY", "prism-test-key") + + model, provider, api_key, api_base = get_llm_provider( + model="prism/deepseek-v4-flash", + custom_llm_provider=None, + api_base=None, + api_key=None, + ) + + assert model == "deepseek-v4-flash" + assert provider == "prism" + assert api_key == "prism-test-key" + assert api_base == "https://api.prisminference.com/v1" + + +def test_prism_provider_keeps_explicit_credentials(monkeypatch: pytest.MonkeyPatch): + from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider + + monkeypatch.setenv("PRISM_API_KEY", "prism-env-key") + + _, provider, api_key, api_base = get_llm_provider( + model="prism/deepseek-v4-flash", + custom_llm_provider=None, + api_base="https://prism.internal.example/v1", + api_key="prism-explicit-key", + ) + + assert provider == "prism" + assert api_key == "prism-explicit-key" + assert api_base == "https://prism.internal.example/v1" + + +def test_prism_model_cost_and_capabilities(): + from litellm.cost_calculator import cost_per_token + + prompt_cost, completion_cost = cost_per_token( + model="prism/deepseek-v4-flash", + prompt_tokens=1_000_000, + completion_tokens=1_000_000, + custom_llm_provider="prism", + ) + model_info = litellm.get_model_info("prism/deepseek-v4-flash") + + assert prompt_cost == pytest.approx(0.14) + assert completion_cost == pytest.approx(0.28) + assert model_info["cache_read_input_token_cost"] == pytest.approx(7e-08) + assert model_info["max_input_tokens"] == 1_000_000 + assert model_info["max_output_tokens"] == 393_216 + assert model_info["supports_function_calling"] is True + assert model_info["supports_native_streaming"] is True + assert model_info["supports_reasoning"] is True + assert model_info["supports_response_schema"] is True