mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-29 01:42:19 +00:00
feat(sail): route Anthropic Messages through Sail's /v1/messages
This commit is contained in:
parent
adbb5b7287
commit
0e94893cd9
4 changed files with 52 additions and 3 deletions
|
|
@ -205,7 +205,7 @@
|
|||
"base_url": "https://api.sailresearch.com/v1",
|
||||
"api_key_env": "SAIL_API_KEY",
|
||||
"api_base_env": "SAIL_API_BASE",
|
||||
"supported_endpoints": ["/v1/chat/completions", "/v1/responses"],
|
||||
"supported_endpoints": ["/v1/chat/completions", "/v1/responses", "/v1/messages"],
|
||||
"param_mappings": {
|
||||
"max_tokens": "max_completion_tokens"
|
||||
}
|
||||
|
|
|
|||
|
|
@ -2034,7 +2034,7 @@
|
|||
"url": "https://docs.litellm.ai/docs/providers/sail",
|
||||
"endpoints": {
|
||||
"chat_completions": true,
|
||||
"messages": false,
|
||||
"messages": true,
|
||||
"responses": true,
|
||||
"embeddings": false,
|
||||
"image_generations": false,
|
||||
|
|
|
|||
|
|
@ -2268,7 +2268,7 @@
|
|||
"url": "https://docs.litellm.ai/docs/providers/sail",
|
||||
"endpoints": {
|
||||
"chat_completions": true,
|
||||
"messages": false,
|
||||
"messages": true,
|
||||
"responses": true,
|
||||
"embeddings": false,
|
||||
"image_generations": false,
|
||||
|
|
|
|||
|
|
@ -15,6 +15,7 @@ from litellm.types.utils import PromptTokensDetailsWrapper, Usage
|
|||
SAIL_BASE_URL = "https://api.sailresearch.com/v1"
|
||||
SAIL_CHAT_COMPLETIONS = f"{SAIL_BASE_URL}/chat/completions"
|
||||
SAIL_RESPONSES = f"{SAIL_BASE_URL}/responses"
|
||||
SAIL_MESSAGES = f"{SAIL_BASE_URL}/messages"
|
||||
|
||||
MODEL = "sail/zai-org/GLM-5.3"
|
||||
|
||||
|
|
@ -76,6 +77,19 @@ def _responses_payload() -> dict:
|
|||
}
|
||||
|
||||
|
||||
def _messages_payload() -> dict:
|
||||
return {
|
||||
"id": "msg_sail",
|
||||
"type": "message",
|
||||
"role": "assistant",
|
||||
"model": "zai-org/GLM-5.3-Flash",
|
||||
"content": [{"type": "text", "text": "sail response"}],
|
||||
"stop_reason": "end_turn",
|
||||
"stop_sequence": None,
|
||||
"usage": {"input_tokens": 2, "output_tokens": 2},
|
||||
}
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _sail_env(monkeypatch: pytest.MonkeyPatch):
|
||||
monkeypatch.setattr(litellm, "disable_aiohttp_transport", True)
|
||||
|
|
@ -236,6 +250,41 @@ class TestSailRequestShape:
|
|||
|
||||
assert not route.called
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.respx()
|
||||
async def test_sail_anthropic_messages_posts_to_messages_endpoint(self, respx_mock: respx.Router):
|
||||
respx_mock.post(SAIL_MESSAGES).respond(json=_messages_payload())
|
||||
|
||||
await litellm.anthropic_messages(
|
||||
model="sail/zai-org/GLM-5.3-Flash",
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
max_tokens=50,
|
||||
)
|
||||
|
||||
assert len(respx_mock.calls) == 1
|
||||
request = respx_mock.calls[0].request
|
||||
assert request.url == SAIL_MESSAGES
|
||||
assert request.headers["Authorization"] == "Bearer sk-sail-test"
|
||||
body = json.loads(request.content)
|
||||
assert body["model"] == "zai-org/GLM-5.3-Flash"
|
||||
assert body["max_tokens"] == 50
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.respx()
|
||||
async def test_sail_anthropic_messages_honors_api_base_env(
|
||||
self, respx_mock: respx.Router, monkeypatch: pytest.MonkeyPatch
|
||||
):
|
||||
monkeypatch.setenv("SAIL_API_BASE", "https://sail.internal.example/v2")
|
||||
route = respx_mock.post("https://sail.internal.example/v2/v1/messages").respond(json=_messages_payload())
|
||||
|
||||
await litellm.anthropic_messages(
|
||||
model="sail/zai-org/GLM-5.3-Flash",
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
max_tokens=50,
|
||||
)
|
||||
|
||||
assert route.called
|
||||
|
||||
|
||||
class TestSailCostTracking:
|
||||
def test_cached_tokens_billed_at_sail_cache_read_rate(self, monkeypatch: pytest.MonkeyPatch):
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue