diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 1fc08c1175f..462f7d2bb40 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -58142,6 +58142,32 @@ "supports_vision": true, "source": "https://docs.sailresearch.com/models" }, + "pinstripes/ps/deepseek-v4-flash": { + "max_tokens": 163840, + "max_input_tokens": 163840, + "max_output_tokens": 163840, + "input_cost_per_token": 1e-07, + "output_cost_per_token": 2e-07, + "litellm_provider": "pinstripes", + "mode": "chat", + "supports_function_calling": true, + "supports_assistant_prefill": true, + "supports_reasoning": true, + "source": "https://pinstripes.io/" + }, + "pinstripes/ps/minimax-m2.7": { + "max_tokens": 1000192, + "max_input_tokens": 1000192, + "max_output_tokens": 1000192, + "input_cost_per_token": 2.55e-07, + "output_cost_per_token": 5.5e-07, + "litellm_provider": "pinstripes", + "mode": "chat", + "supports_function_calling": true, + "supports_assistant_prefill": true, + "supports_reasoning": false, + "source": "https://pinstripes.io/" + }, "darkbloom/gemma-4-26b": { "input_cost_per_token": 3e-08, "litellm_provider": "darkbloom", diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 1fc08c1175f..462f7d2bb40 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -58142,6 +58142,32 @@ "supports_vision": true, "source": "https://docs.sailresearch.com/models" }, + "pinstripes/ps/deepseek-v4-flash": { + "max_tokens": 163840, + "max_input_tokens": 163840, + "max_output_tokens": 163840, + "input_cost_per_token": 1e-07, + "output_cost_per_token": 2e-07, + "litellm_provider": "pinstripes", + "mode": "chat", + "supports_function_calling": true, + "supports_assistant_prefill": true, + "supports_reasoning": true, + "source": "https://pinstripes.io/" + }, + "pinstripes/ps/minimax-m2.7": { + "max_tokens": 1000192, + "max_input_tokens": 1000192, + "max_output_tokens": 1000192, + "input_cost_per_token": 2.55e-07, + "output_cost_per_token": 5.5e-07, + "litellm_provider": "pinstripes", + "mode": "chat", + "supports_function_calling": true, + "supports_assistant_prefill": true, + "supports_reasoning": false, + "source": "https://pinstripes.io/" + }, "darkbloom/gemma-4-26b": { "input_cost_per_token": 3e-08, "litellm_provider": "darkbloom", diff --git a/tests/test_litellm/llms/openai_like/test_sail_provider.py b/tests/test_litellm/llms/openai_like/test_sail_provider.py index 430e720f25b..ac883d9af54 100644 --- a/tests/test_litellm/llms/openai_like/test_sail_provider.py +++ b/tests/test_litellm/llms/openai_like/test_sail_provider.py @@ -1,17 +1,13 @@ -""" -Tests for the Sail (sailresearch.com) JSON-configured provider. - -Each test asserts the outbound HTTP request that litellm would send to Sail, -via a mocked httpx transport, rather than asserting registry contents. -""" +"""Tests for the Sail (sailresearch.com) JSON-configured provider.""" import json -import httpx import pytest import respx import litellm +from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider +from litellm.types.utils import PromptTokensDetailsWrapper, Usage SAIL_BASE_URL = "https://api.sailresearch.com/v1" SAIL_CHAT_COMPLETIONS = f"{SAIL_BASE_URL}/chat/completions" @@ -79,7 +75,7 @@ def _responses_payload() -> dict: @pytest.fixture(autouse=True) def _sail_env(monkeypatch: pytest.MonkeyPatch): - litellm.disable_aiohttp_transport = True + monkeypatch.setattr(litellm, "disable_aiohttp_transport", True) monkeypatch.setenv("SAIL_API_KEY", "sk-sail-test") monkeypatch.delenv("SAIL_API_BASE", raising=False) monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") @@ -101,8 +97,6 @@ class TestSailRequestShape: assert request.url == SAIL_CHAT_COMPLETIONS assert request.headers["Authorization"] == "Bearer sk-sail-test" - from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider - _, provider, _, _ = get_llm_provider( model=MODEL, custom_llm_provider=None, api_base=None, api_key=None ) @@ -229,8 +223,6 @@ class TestSailRequestShape: class TestSailCostTracking: def test_cached_tokens_billed_at_sail_cache_read_rate(self, monkeypatch: pytest.MonkeyPatch): - from litellm.types.utils import PromptTokensDetailsWrapper, Usage - rates = litellm.model_cost[MODEL] prompt_tokens = 1000 cached_tokens = 600