mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-29 01:42:19 +00:00
Register Sail (providers.json, LlmProviders.SAIL, OpenAI-compatible lists, ProviderConfigManager) for chat, Responses and /v1/messages, and add its 12 models to both cost maps with asap, balanced and flex price columns. Sail picks speed and price with metadata.completion_window and rejects service_tier, so the Sail chat and Responses configs translate the tier: default and priority to asap, flex to flex, balanced to balanced, auto to no window. Billing prices the window that was sent. A tier Sail has no window for, or a window or tier set where billing cannot see it (request metadata, extra_body), is a 400 unless drop_params is set. Add balanced to ServiceTier and its _balanced price columns to the model info types, the Rust catalog and the dashboard schema. A transform_extra_body hook on the chat and Responses base configs, which returns extra_body unchanged by default, lets Sail keep the window when a caller also sends extra_body.metadata. Sail is listed in the Add Model form and model picker. Co-authored-by: shrey kharbanda <shrey@berri.ai>
41 lines
1.2 KiB
Python
41 lines
1.2 KiB
Python
import uuid
|
|
from collections.abc import Iterator
|
|
from typing import Final
|
|
|
|
import httpx
|
|
import pytest
|
|
import pytest_asyncio
|
|
import respx
|
|
|
|
import litellm
|
|
from litellm.litellm_core_utils.logging_worker import GLOBAL_LOGGING_WORKER
|
|
from tests.unit.llms.sail.helpers import SAIL_API_BASE, SpendCapture, chat_completion_body
|
|
|
|
|
|
@pytest.fixture
|
|
def sail_env(local_model_cost_map: None, monkeypatch: pytest.MonkeyPatch) -> Iterator[None]:
|
|
monkeypatch.setenv("SAIL_API_KEY", "sail-test-key")
|
|
monkeypatch.delenv("SAIL_API_BASE", raising=False)
|
|
monkeypatch.setattr(
|
|
litellm,
|
|
"disable_aiohttp_transport",
|
|
True,
|
|
)
|
|
litellm.in_memory_llm_clients_cache.flush_cache()
|
|
yield
|
|
litellm.in_memory_llm_clients_cache.flush_cache()
|
|
|
|
|
|
@pytest_asyncio.fixture
|
|
async def spend_capture(monkeypatch: pytest.MonkeyPatch) -> SpendCapture:
|
|
GLOBAL_LOGGING_WORKER.start()
|
|
capture: Final = SpendCapture(call_id=f"sail-{uuid.uuid4()}")
|
|
monkeypatch.setattr(litellm, "callbacks", [capture])
|
|
return capture
|
|
|
|
|
|
@pytest.fixture
|
|
def chat_route(respx_mock: respx.MockRouter) -> respx.Route:
|
|
return respx_mock.post(f"{SAIL_API_BASE}/chat/completions").mock(
|
|
return_value=httpx.Response(200, json=chat_completion_body())
|
|
)
|