mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-11 03:38:38 +00:00
feat(clinepass): add ClinePass provider
ClinePass (the Cline API) is OpenAI-compatible apart from two quirks:
1. Non-streaming completions are nested under a `data` envelope --
`{"data": {"choices": [...]}, "success": true}` -- rather than returning
`choices` at the top level. Against the openai SDK this surfaces as
`r.choices` being None, not as an error. Streaming responses are *not*
enveloped, so SSE needs no special handling.
2. A bare model id is rejected with HTTP 400 "invalid model format. Expected
format: modelType/model", but LiteLLM strips its own `clinepass/` routing
prefix before the request is built, so it has to be restored.
Both are handled in ClinePassConfig, which inherits OpenAIGPTConfig and
overrides only transform_request/async_transform_request (prefix) and
transform_response (unwrap). Unwrapping rebuilds the httpx.Response around the
inner object, so the inherited OpenAI response transform -- including its
`reasoning` -> `reasoning_content` mapping, which Cline populates -- is reused
rather than duplicated.
A response transform is the reason this cannot be a declarative entry in
litellm/llms/openai_like/providers.json: JSON-configured providers are
dispatched to `_complete_custom_openai`, which builds the request via
provider_config but parses the response with convert_to_model_response_object
and never calls provider_config.transform_response. Under that dispatch the
unwrap is unreachable, so ClinePass gets an explicit branch in main.py routing
it to base_llm_http_handler.completion.
`clinepass` is still listed in openai_compatible_providers, which is what
routes an upstream 401 through _map_openai_exception to AuthenticationError;
the explicit dispatch branch precedes that list's catch-all, so membership does
not send it back to the SDK path.
Tests drive litellm.completion()/acompletion() against a mocked transport
rather than calling the transforms directly, so dispatch itself is covered --
a unit test that calls transform_response() proves the function is correct but
not that anything invokes it.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
(cherry picked from commit 86b3486e7cc8834ec9067e45f966b18b7466ec3b)
This commit is contained in:
parent
ad8babae33
commit
2553cbd337
10 changed files with 578 additions and 0 deletions
|
|
@ -2051,6 +2051,9 @@ if TYPE_CHECKING:
|
|||
)
|
||||
from .llms.heroku.chat.transformation import HerokuChatConfig as HerokuChatConfig
|
||||
from .llms.cometapi.chat.transformation import CometAPIConfig as CometAPIConfig
|
||||
from .llms.clinepass.chat.transformation import (
|
||||
ClinePassConfig as ClinePassConfig,
|
||||
)
|
||||
from .llms.azure.chat.gpt_transformation import (
|
||||
AzureOpenAIConfig as AzureOpenAIConfig,
|
||||
)
|
||||
|
|
|
|||
|
|
@ -284,6 +284,7 @@ LLM_CONFIG_NAMES: Final = (
|
|||
"AzureOpenAIAssistantsAPIConfig",
|
||||
"HerokuChatConfig",
|
||||
"CometAPIConfig",
|
||||
"ClinePassConfig",
|
||||
"AzureOpenAIConfig",
|
||||
"AzureOpenAIGPT5Config",
|
||||
"AzureOpenAITextConfig",
|
||||
|
|
@ -1117,6 +1118,7 @@ _LLM_CONFIGS_IMPORT_MAP: Final = {
|
|||
),
|
||||
"HerokuChatConfig": (".llms.heroku.chat.transformation", "HerokuChatConfig"),
|
||||
"CometAPIConfig": (".llms.cometapi.chat.transformation", "CometAPIConfig"),
|
||||
"ClinePassConfig": (".llms.clinepass.chat.transformation", "ClinePassConfig"),
|
||||
"AzureOpenAIConfig": (".llms.azure.chat.gpt_transformation", "AzureOpenAIConfig"),
|
||||
"AzureOpenAIGPT5Config": (
|
||||
".llms.azure.chat.gpt_5_transformation",
|
||||
|
|
|
|||
|
|
@ -782,6 +782,7 @@ LITELLM_CHAT_PROVIDERS: Final = [
|
|||
"lemonade",
|
||||
"docker_model_runner",
|
||||
"amazon_nova",
|
||||
"clinepass",
|
||||
]
|
||||
|
||||
# Resolving these providers runs an OAuth device flow (their provider info IS the login), so any
|
||||
|
|
@ -1041,6 +1042,7 @@ openai_compatible_providers: Final[list] = [
|
|||
"scx-ai",
|
||||
"prism",
|
||||
"sail",
|
||||
"clinepass", # ClinePass (Cline API) - has its own module; listed here for exception mapping
|
||||
]
|
||||
|
||||
OPENAI_AUDIO_TRANSCRIPTION_PROVIDERS: Final = frozenset({"openai"} | frozenset(openai_compatible_providers))
|
||||
|
|
|
|||
|
|
@ -611,6 +611,11 @@ def _get_openai_compatible_provider_info(
|
|||
api_base,
|
||||
dynamic_api_key,
|
||||
) = litellm.PerplexityChatConfig()._get_openai_compatible_provider_info(api_base, api_key)
|
||||
elif custom_llm_provider == "clinepass":
|
||||
(
|
||||
api_base,
|
||||
dynamic_api_key,
|
||||
) = litellm.ClinePassConfig()._get_openai_compatible_provider_info(api_base, api_key)
|
||||
elif custom_llm_provider == "aiohttp_openai":
|
||||
return model, "aiohttp_openai", api_key, api_base
|
||||
elif custom_llm_provider == "anyscale":
|
||||
|
|
|
|||
209
litellm/llms/clinepass/chat/transformation.py
Normal file
209
litellm/llms/clinepass/chat/transformation.py
Normal file
|
|
@ -0,0 +1,209 @@
|
|||
"""
|
||||
Support for ClinePass (the Cline API) `/v1/chat/completions` endpoint.
|
||||
|
||||
ClinePass is OpenAI-compatible apart from two quirks, both handled here:
|
||||
|
||||
1. Non-streaming completions are nested under a ``data`` envelope --
|
||||
``{"data": {"choices": [...]}, "success": true}`` -- rather than returning
|
||||
``choices`` at the top level. Streaming responses are *not* wrapped, so the
|
||||
inherited SSE handling needs no change.
|
||||
2. A bare model id is rejected with HTTP 400 ``invalid model format. Expected
|
||||
format: modelType/model``, but LiteLLM strips its own ``clinepass/`` routing
|
||||
prefix before the request is built, so it has to be restored.
|
||||
|
||||
Documentation: https://docs.cline.bot/
|
||||
"""
|
||||
|
||||
import json
|
||||
from typing import Any, List, Tuple, Union
|
||||
|
||||
import httpx
|
||||
|
||||
from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
||||
from litellm.secret_managers.main import get_secret_str
|
||||
from litellm.types.llms.openai import AllMessageValues
|
||||
from litellm.types.utils import ModelResponse
|
||||
|
||||
from ...openai.chat.gpt_transformation import OpenAIGPTConfig
|
||||
from ..common_utils import ClinePassException
|
||||
|
||||
CLINEPASS_API_BASE = "https://api.cline.bot/api/v1"
|
||||
|
||||
# ClinePass nests the completion under this key on non-streaming responses.
|
||||
CLINEPASS_RESPONSE_ENVELOPE_KEY = "data"
|
||||
|
||||
# The qualifier ClinePass requires on outbound model ids.
|
||||
CLINEPASS_MODEL_PREFIX = "clinepass/"
|
||||
|
||||
# Headers that describe the original byte stream and would be wrong once the
|
||||
# body is rewritten by _unwrap_response_envelope().
|
||||
_BODY_SPECIFIC_HEADERS = ("content-length", "content-encoding")
|
||||
|
||||
|
||||
def _unwrap_response_envelope(raw_response: httpx.Response) -> httpx.Response:
|
||||
"""Strip ClinePass's ``data`` wrapper off a JSON completion body.
|
||||
|
||||
The OpenAI transforms read ``raw_response.json()`` directly, so rebuild the
|
||||
response around the inner object rather than duplicating their bodies here.
|
||||
|
||||
Returns the original response untouched whenever the body does not look like
|
||||
a wrapped completion, so an already-OpenAI-shaped body -- or an error nested
|
||||
under the same key -- is not mistaken for one.
|
||||
"""
|
||||
try:
|
||||
payload = raw_response.json()
|
||||
except (ValueError, httpx.StreamError):
|
||||
# Not a JSON body, or a streaming response that has not been read --
|
||||
# either way there is no envelope to strip.
|
||||
return raw_response
|
||||
|
||||
if not isinstance(payload, dict) or "choices" in payload:
|
||||
return raw_response
|
||||
|
||||
inner = payload.get(CLINEPASS_RESPONSE_ENVELOPE_KEY)
|
||||
if not isinstance(inner, dict) or "choices" not in inner:
|
||||
return raw_response
|
||||
|
||||
headers = {k: v for k, v in raw_response.headers.items() if k.lower() not in _BODY_SPECIFIC_HEADERS}
|
||||
|
||||
return httpx.Response(
|
||||
status_code=raw_response.status_code,
|
||||
headers=headers,
|
||||
content=json.dumps(inner).encode("utf-8"),
|
||||
request=getattr(raw_response, "_request", None),
|
||||
)
|
||||
|
||||
|
||||
def _apply_model_prefix(data: dict) -> dict:
|
||||
"""Restore the ``modelType/model`` qualifier on the outbound model id.
|
||||
|
||||
Only prefix ids that lost their qualifier, so a cross-provider id
|
||||
(``clinepass/openrouter/foo`` -> ``openrouter/foo``) is forwarded unchanged.
|
||||
"""
|
||||
model = data.get("model")
|
||||
if isinstance(model, str) and "/" not in model:
|
||||
data["model"] = f"{CLINEPASS_MODEL_PREFIX}{model}"
|
||||
return data
|
||||
|
||||
|
||||
class ClinePassConfig(OpenAIGPTConfig):
|
||||
"""
|
||||
ClinePass configuration, inheriting the OpenAI chat transforms.
|
||||
|
||||
Overrides only the request/response points where ClinePass diverges; see the
|
||||
module docstring for the two quirks.
|
||||
"""
|
||||
|
||||
def _get_openai_compatible_provider_info(
|
||||
self, api_base: str | None, api_key: str | None
|
||||
) -> Tuple[str | None, str | None]:
|
||||
api_base = api_base or get_secret_str("CLINEPASS_API_BASE") or CLINEPASS_API_BASE
|
||||
dynamic_api_key = api_key or get_secret_str("CLINEPASS_API_KEY")
|
||||
return api_base, dynamic_api_key
|
||||
|
||||
def get_complete_url(
|
||||
self,
|
||||
api_base: str | None,
|
||||
api_key: str | None,
|
||||
model: str,
|
||||
optional_params: dict,
|
||||
litellm_params: dict,
|
||||
stream: bool | None = None,
|
||||
) -> str:
|
||||
if not api_base:
|
||||
api_base = CLINEPASS_API_BASE
|
||||
|
||||
api_base = api_base.rstrip("/")
|
||||
if api_base.endswith("/chat/completions"):
|
||||
return api_base
|
||||
|
||||
return f"{api_base}/chat/completions"
|
||||
|
||||
def map_openai_params(
|
||||
self,
|
||||
non_default_params: dict,
|
||||
optional_params: dict,
|
||||
model: str,
|
||||
drop_params: bool,
|
||||
) -> dict:
|
||||
"""ClinePass takes the legacy ``max_tokens`` spelling only."""
|
||||
mapped_params = super().map_openai_params(
|
||||
non_default_params=non_default_params,
|
||||
optional_params=optional_params,
|
||||
model=model,
|
||||
drop_params=drop_params,
|
||||
)
|
||||
if "max_completion_tokens" in mapped_params:
|
||||
mapped_params["max_tokens"] = mapped_params.pop("max_completion_tokens")
|
||||
return mapped_params
|
||||
|
||||
def transform_request(
|
||||
self,
|
||||
model: str,
|
||||
messages: List[AllMessageValues],
|
||||
optional_params: dict,
|
||||
litellm_params: dict,
|
||||
headers: dict,
|
||||
) -> dict:
|
||||
data = super().transform_request(
|
||||
model=model,
|
||||
messages=messages,
|
||||
optional_params=optional_params,
|
||||
litellm_params=litellm_params,
|
||||
headers=headers,
|
||||
)
|
||||
return _apply_model_prefix(data)
|
||||
|
||||
async def async_transform_request(
|
||||
self,
|
||||
model: str,
|
||||
messages: List[AllMessageValues],
|
||||
optional_params: dict,
|
||||
litellm_params: dict,
|
||||
headers: dict,
|
||||
) -> dict:
|
||||
data = await super().async_transform_request(
|
||||
model=model,
|
||||
messages=messages,
|
||||
optional_params=optional_params,
|
||||
litellm_params=litellm_params,
|
||||
headers=headers,
|
||||
)
|
||||
return _apply_model_prefix(data)
|
||||
|
||||
def transform_response(
|
||||
self,
|
||||
model: str,
|
||||
raw_response: httpx.Response,
|
||||
model_response: ModelResponse,
|
||||
logging_obj: Any,
|
||||
request_data: dict,
|
||||
messages: List[AllMessageValues],
|
||||
optional_params: dict,
|
||||
litellm_params: dict,
|
||||
encoding: Any,
|
||||
api_key: str | None = None,
|
||||
json_mode: bool | None = None,
|
||||
) -> ModelResponse:
|
||||
return super().transform_response(
|
||||
model=model,
|
||||
raw_response=_unwrap_response_envelope(raw_response),
|
||||
model_response=model_response,
|
||||
logging_obj=logging_obj,
|
||||
request_data=request_data,
|
||||
messages=messages,
|
||||
optional_params=optional_params,
|
||||
litellm_params=litellm_params,
|
||||
encoding=encoding,
|
||||
api_key=api_key,
|
||||
json_mode=json_mode,
|
||||
)
|
||||
|
||||
def get_error_class(
|
||||
self, error_message: str, status_code: int, headers: Union[dict, httpx.Headers]
|
||||
) -> BaseLLMException:
|
||||
return ClinePassException(
|
||||
message=error_message,
|
||||
status_code=status_code,
|
||||
headers=headers,
|
||||
)
|
||||
7
litellm/llms/clinepass/common_utils.py
Normal file
7
litellm/llms/clinepass/common_utils.py
Normal file
|
|
@ -0,0 +1,7 @@
|
|||
from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
||||
|
||||
|
||||
class ClinePassException(BaseLLMException):
|
||||
"""ClinePass exception handling class"""
|
||||
|
||||
pass
|
||||
|
|
@ -2448,6 +2448,57 @@ def _complete_aiohttp_openai(
|
|||
)
|
||||
|
||||
|
||||
def _complete_clinepass(ctx: _CompletionDispatchContext) -> _CompletionDispatchResult:
|
||||
acompletion = ctx.acompletion
|
||||
api_base = ctx.api_base
|
||||
api_key = ctx.api_key
|
||||
client = ctx.client
|
||||
custom_llm_provider = ctx.custom_llm_provider
|
||||
headers = ctx.headers
|
||||
litellm_params = ctx.litellm_params
|
||||
logging = ctx.logging
|
||||
messages = ctx.messages
|
||||
model = ctx.model
|
||||
model_response = ctx.model_response
|
||||
optional_params = ctx.optional_params
|
||||
provider_config = ctx.provider_config
|
||||
shared_session = ctx.shared_session
|
||||
stream = ctx.stream
|
||||
timeout = ctx.timeout
|
||||
|
||||
api_key = api_key or get_secret_str("CLINEPASS_API_KEY") or litellm.api_key
|
||||
|
||||
api_base = (
|
||||
api_base or litellm.api_base or get_secret_str("CLINEPASS_API_BASE") or "https://api.cline.bot/api/v1"
|
||||
)
|
||||
|
||||
## COMPLETION CALL
|
||||
response = base_llm_http_handler.completion(
|
||||
model=model,
|
||||
messages=messages,
|
||||
headers=headers,
|
||||
model_response=model_response,
|
||||
api_key=api_key,
|
||||
api_base=api_base,
|
||||
acompletion=acompletion,
|
||||
logging_obj=logging,
|
||||
optional_params=optional_params,
|
||||
litellm_params=litellm_params,
|
||||
shared_session=shared_session,
|
||||
timeout=timeout,
|
||||
client=client,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
encoding=_get_encoding(),
|
||||
stream=stream,
|
||||
provider_config=provider_config,
|
||||
)
|
||||
|
||||
## LOGGING
|
||||
logging.post_call(input=messages, api_key=api_key, original_response=response)
|
||||
|
||||
return response
|
||||
|
||||
|
||||
def _complete_cometapi(ctx: _CompletionDispatchContext) -> _CompletionDispatchResult:
|
||||
acompletion: Final = ctx.acompletion
|
||||
api_base = ctx.api_base
|
||||
|
|
@ -5919,6 +5970,8 @@ def completion(
|
|||
response = _complete_aiohttp_openai(_dispatch_ctx)
|
||||
elif custom_llm_provider == "cometapi":
|
||||
response = _complete_cometapi(_dispatch_ctx)
|
||||
elif custom_llm_provider == "clinepass":
|
||||
response = _complete_clinepass(_dispatch_ctx)
|
||||
elif custom_llm_provider == "minimax":
|
||||
response = _complete_minimax(_dispatch_ctx)
|
||||
elif custom_llm_provider == "hosted_vllm":
|
||||
|
|
|
|||
|
|
@ -4156,6 +4156,7 @@ class LlmProviders(str, Enum):
|
|||
APERTIS = "apertis"
|
||||
NANOGPT = "nano-gpt"
|
||||
POE = "poe"
|
||||
CLINEPASS = "clinepass"
|
||||
CHUTES = "chutes"
|
||||
NEOSANTARA = "neosantara"
|
||||
PARASAIL = "parasail"
|
||||
|
|
|
|||
|
|
@ -8537,6 +8537,7 @@ class ProviderConfigManager:
|
|||
LlmProviders.EDENAI: (litellm.EdenAIChatConfig, False),
|
||||
LlmProviders.FAL_AI: (litellm.FalAIChatConfig, False),
|
||||
LlmProviders.COMETAPI: (lambda: litellm.CometAPIConfig(), False),
|
||||
LlmProviders.CLINEPASS: (lambda: litellm.ClinePassConfig(), False),
|
||||
LlmProviders.DATAROBOT: (lambda: litellm.DataRobotConfig(), False),
|
||||
LlmProviders.GEMINI: (lambda: litellm.GoogleAIStudioGeminiConfig(), False),
|
||||
LlmProviders.AI21: (lambda: litellm.AI21ChatConfig(), False),
|
||||
|
|
|
|||
|
|
@ -0,0 +1,295 @@
|
|||
"""Tests for the ClinePass provider.
|
||||
|
||||
The point of the end-to-end tests here is that they drive ``litellm.completion()``
|
||||
with a mocked transport rather than calling the transforms directly -- a unit test
|
||||
that calls ``transform_response()`` itself proves the function is correct but not
|
||||
that anything invokes it, which is exactly how the envelope unwrap was previously
|
||||
shipped as dead code.
|
||||
"""
|
||||
|
||||
import json
|
||||
from unittest.mock import patch
|
||||
|
||||
import httpx
|
||||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.llms.clinepass.chat.transformation import (
|
||||
ClinePassConfig,
|
||||
_apply_model_prefix,
|
||||
_unwrap_response_envelope,
|
||||
)
|
||||
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler
|
||||
from litellm.types.utils import LlmProviders
|
||||
from litellm.utils import ProviderConfigManager
|
||||
|
||||
API_KEY = "sk-clinepass-test-not-real"
|
||||
|
||||
ENVELOPED_COMPLETION = {
|
||||
"success": True,
|
||||
"data": {
|
||||
"id": "chatcmpl-test",
|
||||
"object": "chat.completion",
|
||||
"created": 1,
|
||||
"model": "clinepass/deepseek-v4-flash",
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"finish_reason": "stop",
|
||||
"message": {
|
||||
"role": "assistant",
|
||||
"content": "pong",
|
||||
"reasoning": "the user asked for pong",
|
||||
},
|
||||
}
|
||||
],
|
||||
"usage": {"prompt_tokens": 5, "completion_tokens": 2, "total_tokens": 7},
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def _response(payload: dict, url: str = "https://api.cline.bot/api/v1/chat/completions") -> httpx.Response:
|
||||
return httpx.Response(200, json=payload, request=httpx.Request("POST", url))
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _clinepass_env(monkeypatch):
|
||||
monkeypatch.setenv("CLINEPASS_API_KEY", API_KEY)
|
||||
monkeypatch.delenv("CLINEPASS_API_BASE", raising=False)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Registration / routing
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_get_llm_provider_resolves_clinepass():
|
||||
model, provider, api_key, api_base = litellm.get_llm_provider(model="clinepass/deepseek-v4-flash")
|
||||
assert model == "deepseek-v4-flash"
|
||||
assert provider == "clinepass"
|
||||
assert api_key == API_KEY
|
||||
assert api_base == "https://api.cline.bot/api/v1"
|
||||
|
||||
|
||||
def test_provider_config_manager_returns_clinepass_config():
|
||||
config = ProviderConfigManager.get_provider_chat_config(model="deepseek-v4-flash", provider=LlmProviders.CLINEPASS)
|
||||
assert isinstance(config, ClinePassConfig)
|
||||
|
||||
|
||||
def test_clinepass_is_not_a_json_configured_provider():
|
||||
"""ClinePass needs a response transform, which the JSON provider system's
|
||||
OpenAI-SDK dispatch path never invokes. Guard against it drifting back."""
|
||||
from litellm.llms.openai_like.json_loader import JSONProviderRegistry
|
||||
|
||||
assert not JSONProviderRegistry.exists("clinepass")
|
||||
|
||||
|
||||
def test_clinepass_stays_in_openai_compatible_providers():
|
||||
"""Membership drives `_map_openai_exception`, so dropping it silently
|
||||
downgrades a 401 to APIConnectionError. The explicit dispatch branch in
|
||||
main.py precedes the openai_compatible_providers catch-all, so being listed
|
||||
here does NOT route ClinePass to the OpenAI SDK path."""
|
||||
assert "clinepass" in litellm.openai_compatible_providers
|
||||
|
||||
|
||||
def test_api_base_env_override(monkeypatch):
|
||||
monkeypatch.setenv("CLINEPASS_API_BASE", "https://proxy.internal/api/v1")
|
||||
_, _, _, api_base = litellm.get_llm_provider(model="clinepass/deepseek-v4-flash")
|
||||
assert api_base == "https://proxy.internal/api/v1"
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"api_base,expected",
|
||||
[
|
||||
(None, "https://api.cline.bot/api/v1/chat/completions"),
|
||||
("https://api.cline.bot/api/v1", "https://api.cline.bot/api/v1/chat/completions"),
|
||||
("https://api.cline.bot/api/v1/", "https://api.cline.bot/api/v1/chat/completions"),
|
||||
(
|
||||
"https://api.cline.bot/api/v1/chat/completions",
|
||||
"https://api.cline.bot/api/v1/chat/completions",
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_get_complete_url(api_base, expected):
|
||||
url = ClinePassConfig().get_complete_url(
|
||||
api_base=api_base, api_key=API_KEY, model="deepseek-v4-flash", optional_params={}, litellm_params={}
|
||||
)
|
||||
assert url == expected
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Model prefix
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_model_prefix_restored_on_bare_id():
|
||||
assert _apply_model_prefix({"model": "deepseek-v4-flash"})["model"] == "clinepass/deepseek-v4-flash"
|
||||
|
||||
|
||||
def test_model_prefix_left_alone_when_qualifier_present():
|
||||
"""`clinepass/openrouter/foo` arrives here as `openrouter/foo` and must pass through."""
|
||||
assert _apply_model_prefix({"model": "openrouter/foo"})["model"] == "openrouter/foo"
|
||||
|
||||
|
||||
def test_model_prefix_ignores_missing_model():
|
||||
assert _apply_model_prefix({}) == {}
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Response envelope
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_unwrap_envelope_extracts_inner_completion():
|
||||
unwrapped = _unwrap_response_envelope(_response(ENVELOPED_COMPLETION))
|
||||
assert unwrapped.json() == ENVELOPED_COMPLETION["data"]
|
||||
|
||||
|
||||
def test_unwrap_envelope_content_length_describes_the_new_body():
|
||||
"""The original content-length describes the enveloped bytes and must not be
|
||||
carried over; httpx recomputes a correct one for the rewritten body."""
|
||||
raw = _response(ENVELOPED_COMPLETION)
|
||||
unwrapped = _unwrap_response_envelope(raw)
|
||||
assert unwrapped.headers["content-length"] != raw.headers["content-length"]
|
||||
assert int(unwrapped.headers["content-length"]) == len(unwrapped.content)
|
||||
|
||||
|
||||
def test_unwrap_envelope_passes_through_openai_shaped_body():
|
||||
payload = ENVELOPED_COMPLETION["data"]
|
||||
assert _unwrap_response_envelope(_response(payload)).json() == payload
|
||||
|
||||
|
||||
def test_unwrap_envelope_passes_through_error_nested_under_same_key():
|
||||
"""An error under `data` has no `choices` and must not be mistaken for a completion."""
|
||||
payload = {"success": False, "data": {"message": "bad model"}}
|
||||
assert _unwrap_response_envelope(_response(payload)).json() == payload
|
||||
|
||||
|
||||
def test_unwrap_envelope_passes_through_non_json_body():
|
||||
raw = httpx.Response(
|
||||
200, content=b"not json", request=httpx.Request("POST", "https://api.cline.bot/api/v1/chat/completions")
|
||||
)
|
||||
assert _unwrap_response_envelope(raw) is raw
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Parameter mapping
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_max_completion_tokens_mapped_to_max_tokens():
|
||||
mapped = ClinePassConfig().map_openai_params(
|
||||
non_default_params={"max_completion_tokens": 4000},
|
||||
optional_params={},
|
||||
model="deepseek-v4-flash",
|
||||
drop_params=False,
|
||||
)
|
||||
assert mapped == {"max_tokens": 4000}
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# End-to-end through litellm.completion() -- these are the load-bearing ones
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_completion_unwraps_envelope_and_prefixes_model():
|
||||
captured = {}
|
||||
|
||||
def fake_post(self, url, *args, **kwargs):
|
||||
captured["url"] = str(url)
|
||||
captured["body"] = json.loads(kwargs["data"])
|
||||
return _response(ENVELOPED_COMPLETION)
|
||||
|
||||
with patch.object(HTTPHandler, "post", fake_post):
|
||||
response = litellm.completion(
|
||||
model="clinepass/deepseek-v4-flash",
|
||||
messages=[{"role": "user", "content": "ping"}],
|
||||
max_tokens=4000,
|
||||
)
|
||||
|
||||
assert captured["url"] == "https://api.cline.bot/api/v1/chat/completions"
|
||||
assert captured["body"]["model"] == "clinepass/deepseek-v4-flash"
|
||||
assert response.choices[0].message.content == "pong"
|
||||
assert response.choices[0].message.reasoning_content == "the user asked for pong"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_acompletion_unwraps_envelope_and_prefixes_model():
|
||||
captured = {}
|
||||
|
||||
async def fake_post(self, url, *args, **kwargs):
|
||||
captured["url"] = str(url)
|
||||
captured["body"] = json.loads(kwargs["data"])
|
||||
return _response(ENVELOPED_COMPLETION)
|
||||
|
||||
with patch.object(AsyncHTTPHandler, "post", fake_post):
|
||||
response = await litellm.acompletion(
|
||||
model="clinepass/deepseek-v4-flash",
|
||||
messages=[{"role": "user", "content": "ping"}],
|
||||
max_tokens=4000,
|
||||
)
|
||||
|
||||
assert captured["url"] == "https://api.cline.bot/api/v1/chat/completions"
|
||||
assert captured["body"]["model"] == "clinepass/deepseek-v4-flash"
|
||||
assert response.choices[0].message.content == "pong"
|
||||
|
||||
|
||||
def test_completion_streaming_is_not_unwrapped():
|
||||
"""ClinePass does NOT wrap SSE chunks -- they are already OpenAI-shaped."""
|
||||
chunks = [
|
||||
{
|
||||
"id": "chatcmpl-test",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 1,
|
||||
"model": "clinepass/deepseek-v4-flash",
|
||||
"choices": [{"index": 0, "delta": {"content": piece}, "finish_reason": None}],
|
||||
}
|
||||
for piece in ["one ", "two ", "three"]
|
||||
]
|
||||
body = "".join(f"data: {json.dumps(c)}\n\n" for c in chunks) + "data: [DONE]\n\n"
|
||||
|
||||
def fake_post(self, url, *args, **kwargs):
|
||||
return httpx.Response(
|
||||
200,
|
||||
content=body.encode(),
|
||||
headers={"content-type": "text/event-stream"},
|
||||
request=httpx.Request("POST", str(url)),
|
||||
)
|
||||
|
||||
with patch.object(HTTPHandler, "post", fake_post):
|
||||
stream = litellm.completion(
|
||||
model="clinepass/deepseek-v4-flash",
|
||||
messages=[{"role": "user", "content": "count"}],
|
||||
max_tokens=4000,
|
||||
stream=True,
|
||||
)
|
||||
text = "".join(c.choices[0].delta.content or "" for c in stream if c.choices)
|
||||
|
||||
assert text == "one two three"
|
||||
|
||||
|
||||
def test_upstream_401_maps_to_authentication_error():
|
||||
"""ClinePass answers a bad key with HTTP 401; that must not be flattened
|
||||
into a generic APIConnectionError."""
|
||||
from litellm.exceptions import AuthenticationError
|
||||
|
||||
def fake_post(self, url, *args, **kwargs):
|
||||
raise httpx.HTTPStatusError(
|
||||
"Unauthorized",
|
||||
request=httpx.Request("POST", str(url)),
|
||||
response=httpx.Response(
|
||||
401,
|
||||
json={"error": "Unauthorized"},
|
||||
request=httpx.Request("POST", str(url)),
|
||||
),
|
||||
)
|
||||
|
||||
with patch.object(HTTPHandler, "post", fake_post):
|
||||
with pytest.raises(AuthenticationError) as excinfo:
|
||||
litellm.completion(
|
||||
model="clinepass/deepseek-v4-flash",
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
max_tokens=100,
|
||||
)
|
||||
|
||||
assert excinfo.value.status_code == 401
|
||||
Loading…
Add table
Reference in a new issue