feat(clinepass): add ClinePass provider

ClinePass (the Cline API) is OpenAI-compatible apart from two quirks:

1. Non-streaming completions are nested under a `data` envelope --
   `{"data": {"choices": [...]}, "success": true}` -- rather than returning
   `choices` at the top level. Against the openai SDK this surfaces as
   `r.choices` being None, not as an error. Streaming responses are *not*
   enveloped, so SSE needs no special handling.
2. A bare model id is rejected with HTTP 400 "invalid model format. Expected
   format: modelType/model", but LiteLLM strips its own `clinepass/` routing
   prefix before the request is built, so it has to be restored.

Both are handled in ClinePassConfig, which inherits OpenAIGPTConfig and
overrides only transform_request/async_transform_request (prefix) and
transform_response (unwrap). Unwrapping rebuilds the httpx.Response around the
inner object, so the inherited OpenAI response transform -- including its
`reasoning` -> `reasoning_content` mapping, which Cline populates -- is reused
rather than duplicated.

A response transform is the reason this cannot be a declarative entry in
litellm/llms/openai_like/providers.json: JSON-configured providers are
dispatched to `_complete_custom_openai`, which builds the request via
provider_config but parses the response with convert_to_model_response_object
and never calls provider_config.transform_response. Under that dispatch the
unwrap is unreachable, so ClinePass gets an explicit branch in main.py routing
it to base_llm_http_handler.completion.

`clinepass` is still listed in openai_compatible_providers, which is what
routes an upstream 401 through _map_openai_exception to AuthenticationError;
the explicit dispatch branch precedes that list's catch-all, so membership does
not send it back to the SDK path.

Tests drive litellm.completion()/acompletion() against a mocked transport
rather than calling the transforms directly, so dispatch itself is covered --
a unit test that calls transform_response() proves the function is correct but
not that anything invokes it.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
(cherry picked from commit 86b3486e7cc8834ec9067e45f966b18b7466ec3b)
This commit is contained in:
Daniel JB Clark 2026-08-21 22:36:24 -04:00
parent ad8babae33
commit 2553cbd337
No known key found for this signature in database
10 changed files with 578 additions and 0 deletions

View file

@ -2051,6 +2051,9 @@ if TYPE_CHECKING:
)
from .llms.heroku.chat.transformation import HerokuChatConfig as HerokuChatConfig
from .llms.cometapi.chat.transformation import CometAPIConfig as CometAPIConfig
from .llms.clinepass.chat.transformation import (
ClinePassConfig as ClinePassConfig,
)
from .llms.azure.chat.gpt_transformation import (
AzureOpenAIConfig as AzureOpenAIConfig,
)

View file

@ -284,6 +284,7 @@ LLM_CONFIG_NAMES: Final = (
"AzureOpenAIAssistantsAPIConfig",
"HerokuChatConfig",
"CometAPIConfig",
"ClinePassConfig",
"AzureOpenAIConfig",
"AzureOpenAIGPT5Config",
"AzureOpenAITextConfig",
@ -1117,6 +1118,7 @@ _LLM_CONFIGS_IMPORT_MAP: Final = {
),
"HerokuChatConfig": (".llms.heroku.chat.transformation", "HerokuChatConfig"),
"CometAPIConfig": (".llms.cometapi.chat.transformation", "CometAPIConfig"),
"ClinePassConfig": (".llms.clinepass.chat.transformation", "ClinePassConfig"),
"AzureOpenAIConfig": (".llms.azure.chat.gpt_transformation", "AzureOpenAIConfig"),
"AzureOpenAIGPT5Config": (
".llms.azure.chat.gpt_5_transformation",

View file

@ -782,6 +782,7 @@ LITELLM_CHAT_PROVIDERS: Final = [
"lemonade",
"docker_model_runner",
"amazon_nova",
"clinepass",
]
# Resolving these providers runs an OAuth device flow (their provider info IS the login), so any
@ -1041,6 +1042,7 @@ openai_compatible_providers: Final[list] = [
"scx-ai",
"prism",
"sail",
"clinepass", # ClinePass (Cline API) - has its own module; listed here for exception mapping
]
OPENAI_AUDIO_TRANSCRIPTION_PROVIDERS: Final = frozenset({"openai"} | frozenset(openai_compatible_providers))

View file

@ -611,6 +611,11 @@ def _get_openai_compatible_provider_info(
api_base,
dynamic_api_key,
) = litellm.PerplexityChatConfig()._get_openai_compatible_provider_info(api_base, api_key)
elif custom_llm_provider == "clinepass":
(
api_base,
dynamic_api_key,
) = litellm.ClinePassConfig()._get_openai_compatible_provider_info(api_base, api_key)
elif custom_llm_provider == "aiohttp_openai":
return model, "aiohttp_openai", api_key, api_base
elif custom_llm_provider == "anyscale":

View file

@ -0,0 +1,209 @@
"""
Support for ClinePass (the Cline API) `/v1/chat/completions` endpoint.
ClinePass is OpenAI-compatible apart from two quirks, both handled here:
1. Non-streaming completions are nested under a ``data`` envelope --
``{"data": {"choices": [...]}, "success": true}`` -- rather than returning
``choices`` at the top level. Streaming responses are *not* wrapped, so the
inherited SSE handling needs no change.
2. A bare model id is rejected with HTTP 400 ``invalid model format. Expected
format: modelType/model``, but LiteLLM strips its own ``clinepass/`` routing
prefix before the request is built, so it has to be restored.
Documentation: https://docs.cline.bot/
"""
import json
from typing import Any, List, Tuple, Union
import httpx
from litellm.llms.base_llm.chat.transformation import BaseLLMException
from litellm.secret_managers.main import get_secret_str
from litellm.types.llms.openai import AllMessageValues
from litellm.types.utils import ModelResponse
from ...openai.chat.gpt_transformation import OpenAIGPTConfig
from ..common_utils import ClinePassException
CLINEPASS_API_BASE = "https://api.cline.bot/api/v1"
# ClinePass nests the completion under this key on non-streaming responses.
CLINEPASS_RESPONSE_ENVELOPE_KEY = "data"
# The qualifier ClinePass requires on outbound model ids.
CLINEPASS_MODEL_PREFIX = "clinepass/"
# Headers that describe the original byte stream and would be wrong once the
# body is rewritten by _unwrap_response_envelope().
_BODY_SPECIFIC_HEADERS = ("content-length", "content-encoding")
def _unwrap_response_envelope(raw_response: httpx.Response) -> httpx.Response:
"""Strip ClinePass's ``data`` wrapper off a JSON completion body.
The OpenAI transforms read ``raw_response.json()`` directly, so rebuild the
response around the inner object rather than duplicating their bodies here.
Returns the original response untouched whenever the body does not look like
a wrapped completion, so an already-OpenAI-shaped body -- or an error nested
under the same key -- is not mistaken for one.
"""
try:
payload = raw_response.json()
except (ValueError, httpx.StreamError):
# Not a JSON body, or a streaming response that has not been read --
# either way there is no envelope to strip.
return raw_response
if not isinstance(payload, dict) or "choices" in payload:
return raw_response
inner = payload.get(CLINEPASS_RESPONSE_ENVELOPE_KEY)
if not isinstance(inner, dict) or "choices" not in inner:
return raw_response
headers = {k: v for k, v in raw_response.headers.items() if k.lower() not in _BODY_SPECIFIC_HEADERS}
return httpx.Response(
status_code=raw_response.status_code,
headers=headers,
content=json.dumps(inner).encode("utf-8"),
request=getattr(raw_response, "_request", None),
)
def _apply_model_prefix(data: dict) -> dict:
"""Restore the ``modelType/model`` qualifier on the outbound model id.
Only prefix ids that lost their qualifier, so a cross-provider id
(``clinepass/openrouter/foo`` -> ``openrouter/foo``) is forwarded unchanged.
"""
model = data.get("model")
if isinstance(model, str) and "/" not in model:
data["model"] = f"{CLINEPASS_MODEL_PREFIX}{model}"
return data
class ClinePassConfig(OpenAIGPTConfig):
"""
ClinePass configuration, inheriting the OpenAI chat transforms.
Overrides only the request/response points where ClinePass diverges; see the
module docstring for the two quirks.
"""
def _get_openai_compatible_provider_info(
self, api_base: str | None, api_key: str | None
) -> Tuple[str | None, str | None]:
api_base = api_base or get_secret_str("CLINEPASS_API_BASE") or CLINEPASS_API_BASE
dynamic_api_key = api_key or get_secret_str("CLINEPASS_API_KEY")
return api_base, dynamic_api_key
def get_complete_url(
self,
api_base: str | None,
api_key: str | None,
model: str,
optional_params: dict,
litellm_params: dict,
stream: bool | None = None,
) -> str:
if not api_base:
api_base = CLINEPASS_API_BASE
api_base = api_base.rstrip("/")
if api_base.endswith("/chat/completions"):
return api_base
return f"{api_base}/chat/completions"
def map_openai_params(
self,
non_default_params: dict,
optional_params: dict,
model: str,
drop_params: bool,
) -> dict:
"""ClinePass takes the legacy ``max_tokens`` spelling only."""
mapped_params = super().map_openai_params(
non_default_params=non_default_params,
optional_params=optional_params,
model=model,
drop_params=drop_params,
)
if "max_completion_tokens" in mapped_params:
mapped_params["max_tokens"] = mapped_params.pop("max_completion_tokens")
return mapped_params
def transform_request(
self,
model: str,
messages: List[AllMessageValues],
optional_params: dict,
litellm_params: dict,
headers: dict,
) -> dict:
data = super().transform_request(
model=model,
messages=messages,
optional_params=optional_params,
litellm_params=litellm_params,
headers=headers,
)
return _apply_model_prefix(data)
async def async_transform_request(
self,
model: str,
messages: List[AllMessageValues],
optional_params: dict,
litellm_params: dict,
headers: dict,
) -> dict:
data = await super().async_transform_request(
model=model,
messages=messages,
optional_params=optional_params,
litellm_params=litellm_params,
headers=headers,
)
return _apply_model_prefix(data)
def transform_response(
self,
model: str,
raw_response: httpx.Response,
model_response: ModelResponse,
logging_obj: Any,
request_data: dict,
messages: List[AllMessageValues],
optional_params: dict,
litellm_params: dict,
encoding: Any,
api_key: str | None = None,
json_mode: bool | None = None,
) -> ModelResponse:
return super().transform_response(
model=model,
raw_response=_unwrap_response_envelope(raw_response),
model_response=model_response,
logging_obj=logging_obj,
request_data=request_data,
messages=messages,
optional_params=optional_params,
litellm_params=litellm_params,
encoding=encoding,
api_key=api_key,
json_mode=json_mode,
)
def get_error_class(
self, error_message: str, status_code: int, headers: Union[dict, httpx.Headers]
) -> BaseLLMException:
return ClinePassException(
message=error_message,
status_code=status_code,
headers=headers,
)

View file

@ -0,0 +1,7 @@
from litellm.llms.base_llm.chat.transformation import BaseLLMException
class ClinePassException(BaseLLMException):
"""ClinePass exception handling class"""
pass

View file

@ -2448,6 +2448,57 @@ def _complete_aiohttp_openai(
)
def _complete_clinepass(ctx: _CompletionDispatchContext) -> _CompletionDispatchResult:
acompletion = ctx.acompletion
api_base = ctx.api_base
api_key = ctx.api_key
client = ctx.client
custom_llm_provider = ctx.custom_llm_provider
headers = ctx.headers
litellm_params = ctx.litellm_params
logging = ctx.logging
messages = ctx.messages
model = ctx.model
model_response = ctx.model_response
optional_params = ctx.optional_params
provider_config = ctx.provider_config
shared_session = ctx.shared_session
stream = ctx.stream
timeout = ctx.timeout
api_key = api_key or get_secret_str("CLINEPASS_API_KEY") or litellm.api_key
api_base = (
api_base or litellm.api_base or get_secret_str("CLINEPASS_API_BASE") or "https://api.cline.bot/api/v1"
)
## COMPLETION CALL
response = base_llm_http_handler.completion(
model=model,
messages=messages,
headers=headers,
model_response=model_response,
api_key=api_key,
api_base=api_base,
acompletion=acompletion,
logging_obj=logging,
optional_params=optional_params,
litellm_params=litellm_params,
shared_session=shared_session,
timeout=timeout,
client=client,
custom_llm_provider=custom_llm_provider,
encoding=_get_encoding(),
stream=stream,
provider_config=provider_config,
)
## LOGGING
logging.post_call(input=messages, api_key=api_key, original_response=response)
return response
def _complete_cometapi(ctx: _CompletionDispatchContext) -> _CompletionDispatchResult:
acompletion: Final = ctx.acompletion
api_base = ctx.api_base
@ -5919,6 +5970,8 @@ def completion(
response = _complete_aiohttp_openai(_dispatch_ctx)
elif custom_llm_provider == "cometapi":
response = _complete_cometapi(_dispatch_ctx)
elif custom_llm_provider == "clinepass":
response = _complete_clinepass(_dispatch_ctx)
elif custom_llm_provider == "minimax":
response = _complete_minimax(_dispatch_ctx)
elif custom_llm_provider == "hosted_vllm":

View file

@ -4156,6 +4156,7 @@ class LlmProviders(str, Enum):
APERTIS = "apertis"
NANOGPT = "nano-gpt"
POE = "poe"
CLINEPASS = "clinepass"
CHUTES = "chutes"
NEOSANTARA = "neosantara"
PARASAIL = "parasail"

View file

@ -8537,6 +8537,7 @@ class ProviderConfigManager:
LlmProviders.EDENAI: (litellm.EdenAIChatConfig, False),
LlmProviders.FAL_AI: (litellm.FalAIChatConfig, False),
LlmProviders.COMETAPI: (lambda: litellm.CometAPIConfig(), False),
LlmProviders.CLINEPASS: (lambda: litellm.ClinePassConfig(), False),
LlmProviders.DATAROBOT: (lambda: litellm.DataRobotConfig(), False),
LlmProviders.GEMINI: (lambda: litellm.GoogleAIStudioGeminiConfig(), False),
LlmProviders.AI21: (lambda: litellm.AI21ChatConfig(), False),

View file

@ -0,0 +1,295 @@
"""Tests for the ClinePass provider.
The point of the end-to-end tests here is that they drive ``litellm.completion()``
with a mocked transport rather than calling the transforms directly -- a unit test
that calls ``transform_response()`` itself proves the function is correct but not
that anything invokes it, which is exactly how the envelope unwrap was previously
shipped as dead code.
"""
import json
from unittest.mock import patch
import httpx
import pytest
import litellm
from litellm.llms.clinepass.chat.transformation import (
ClinePassConfig,
_apply_model_prefix,
_unwrap_response_envelope,
)
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler
from litellm.types.utils import LlmProviders
from litellm.utils import ProviderConfigManager
API_KEY = "sk-clinepass-test-not-real"
ENVELOPED_COMPLETION = {
"success": True,
"data": {
"id": "chatcmpl-test",
"object": "chat.completion",
"created": 1,
"model": "clinepass/deepseek-v4-flash",
"choices": [
{
"index": 0,
"finish_reason": "stop",
"message": {
"role": "assistant",
"content": "pong",
"reasoning": "the user asked for pong",
},
}
],
"usage": {"prompt_tokens": 5, "completion_tokens": 2, "total_tokens": 7},
},
}
def _response(payload: dict, url: str = "https://api.cline.bot/api/v1/chat/completions") -> httpx.Response:
return httpx.Response(200, json=payload, request=httpx.Request("POST", url))
@pytest.fixture(autouse=True)
def _clinepass_env(monkeypatch):
monkeypatch.setenv("CLINEPASS_API_KEY", API_KEY)
monkeypatch.delenv("CLINEPASS_API_BASE", raising=False)
# --------------------------------------------------------------------------
# Registration / routing
# --------------------------------------------------------------------------
def test_get_llm_provider_resolves_clinepass():
model, provider, api_key, api_base = litellm.get_llm_provider(model="clinepass/deepseek-v4-flash")
assert model == "deepseek-v4-flash"
assert provider == "clinepass"
assert api_key == API_KEY
assert api_base == "https://api.cline.bot/api/v1"
def test_provider_config_manager_returns_clinepass_config():
config = ProviderConfigManager.get_provider_chat_config(model="deepseek-v4-flash", provider=LlmProviders.CLINEPASS)
assert isinstance(config, ClinePassConfig)
def test_clinepass_is_not_a_json_configured_provider():
"""ClinePass needs a response transform, which the JSON provider system's
OpenAI-SDK dispatch path never invokes. Guard against it drifting back."""
from litellm.llms.openai_like.json_loader import JSONProviderRegistry
assert not JSONProviderRegistry.exists("clinepass")
def test_clinepass_stays_in_openai_compatible_providers():
"""Membership drives `_map_openai_exception`, so dropping it silently
downgrades a 401 to APIConnectionError. The explicit dispatch branch in
main.py precedes the openai_compatible_providers catch-all, so being listed
here does NOT route ClinePass to the OpenAI SDK path."""
assert "clinepass" in litellm.openai_compatible_providers
def test_api_base_env_override(monkeypatch):
monkeypatch.setenv("CLINEPASS_API_BASE", "https://proxy.internal/api/v1")
_, _, _, api_base = litellm.get_llm_provider(model="clinepass/deepseek-v4-flash")
assert api_base == "https://proxy.internal/api/v1"
@pytest.mark.parametrize(
"api_base,expected",
[
(None, "https://api.cline.bot/api/v1/chat/completions"),
("https://api.cline.bot/api/v1", "https://api.cline.bot/api/v1/chat/completions"),
("https://api.cline.bot/api/v1/", "https://api.cline.bot/api/v1/chat/completions"),
(
"https://api.cline.bot/api/v1/chat/completions",
"https://api.cline.bot/api/v1/chat/completions",
),
],
)
def test_get_complete_url(api_base, expected):
url = ClinePassConfig().get_complete_url(
api_base=api_base, api_key=API_KEY, model="deepseek-v4-flash", optional_params={}, litellm_params={}
)
assert url == expected
# --------------------------------------------------------------------------
# Model prefix
# --------------------------------------------------------------------------
def test_model_prefix_restored_on_bare_id():
assert _apply_model_prefix({"model": "deepseek-v4-flash"})["model"] == "clinepass/deepseek-v4-flash"
def test_model_prefix_left_alone_when_qualifier_present():
"""`clinepass/openrouter/foo` arrives here as `openrouter/foo` and must pass through."""
assert _apply_model_prefix({"model": "openrouter/foo"})["model"] == "openrouter/foo"
def test_model_prefix_ignores_missing_model():
assert _apply_model_prefix({}) == {}
# --------------------------------------------------------------------------
# Response envelope
# --------------------------------------------------------------------------
def test_unwrap_envelope_extracts_inner_completion():
unwrapped = _unwrap_response_envelope(_response(ENVELOPED_COMPLETION))
assert unwrapped.json() == ENVELOPED_COMPLETION["data"]
def test_unwrap_envelope_content_length_describes_the_new_body():
"""The original content-length describes the enveloped bytes and must not be
carried over; httpx recomputes a correct one for the rewritten body."""
raw = _response(ENVELOPED_COMPLETION)
unwrapped = _unwrap_response_envelope(raw)
assert unwrapped.headers["content-length"] != raw.headers["content-length"]
assert int(unwrapped.headers["content-length"]) == len(unwrapped.content)
def test_unwrap_envelope_passes_through_openai_shaped_body():
payload = ENVELOPED_COMPLETION["data"]
assert _unwrap_response_envelope(_response(payload)).json() == payload
def test_unwrap_envelope_passes_through_error_nested_under_same_key():
"""An error under `data` has no `choices` and must not be mistaken for a completion."""
payload = {"success": False, "data": {"message": "bad model"}}
assert _unwrap_response_envelope(_response(payload)).json() == payload
def test_unwrap_envelope_passes_through_non_json_body():
raw = httpx.Response(
200, content=b"not json", request=httpx.Request("POST", "https://api.cline.bot/api/v1/chat/completions")
)
assert _unwrap_response_envelope(raw) is raw
# --------------------------------------------------------------------------
# Parameter mapping
# --------------------------------------------------------------------------
def test_max_completion_tokens_mapped_to_max_tokens():
mapped = ClinePassConfig().map_openai_params(
non_default_params={"max_completion_tokens": 4000},
optional_params={},
model="deepseek-v4-flash",
drop_params=False,
)
assert mapped == {"max_tokens": 4000}
# --------------------------------------------------------------------------
# End-to-end through litellm.completion() -- these are the load-bearing ones
# --------------------------------------------------------------------------
def test_completion_unwraps_envelope_and_prefixes_model():
captured = {}
def fake_post(self, url, *args, **kwargs):
captured["url"] = str(url)
captured["body"] = json.loads(kwargs["data"])
return _response(ENVELOPED_COMPLETION)
with patch.object(HTTPHandler, "post", fake_post):
response = litellm.completion(
model="clinepass/deepseek-v4-flash",
messages=[{"role": "user", "content": "ping"}],
max_tokens=4000,
)
assert captured["url"] == "https://api.cline.bot/api/v1/chat/completions"
assert captured["body"]["model"] == "clinepass/deepseek-v4-flash"
assert response.choices[0].message.content == "pong"
assert response.choices[0].message.reasoning_content == "the user asked for pong"
@pytest.mark.asyncio
async def test_acompletion_unwraps_envelope_and_prefixes_model():
captured = {}
async def fake_post(self, url, *args, **kwargs):
captured["url"] = str(url)
captured["body"] = json.loads(kwargs["data"])
return _response(ENVELOPED_COMPLETION)
with patch.object(AsyncHTTPHandler, "post", fake_post):
response = await litellm.acompletion(
model="clinepass/deepseek-v4-flash",
messages=[{"role": "user", "content": "ping"}],
max_tokens=4000,
)
assert captured["url"] == "https://api.cline.bot/api/v1/chat/completions"
assert captured["body"]["model"] == "clinepass/deepseek-v4-flash"
assert response.choices[0].message.content == "pong"
def test_completion_streaming_is_not_unwrapped():
"""ClinePass does NOT wrap SSE chunks -- they are already OpenAI-shaped."""
chunks = [
{
"id": "chatcmpl-test",
"object": "chat.completion.chunk",
"created": 1,
"model": "clinepass/deepseek-v4-flash",
"choices": [{"index": 0, "delta": {"content": piece}, "finish_reason": None}],
}
for piece in ["one ", "two ", "three"]
]
body = "".join(f"data: {json.dumps(c)}\n\n" for c in chunks) + "data: [DONE]\n\n"
def fake_post(self, url, *args, **kwargs):
return httpx.Response(
200,
content=body.encode(),
headers={"content-type": "text/event-stream"},
request=httpx.Request("POST", str(url)),
)
with patch.object(HTTPHandler, "post", fake_post):
stream = litellm.completion(
model="clinepass/deepseek-v4-flash",
messages=[{"role": "user", "content": "count"}],
max_tokens=4000,
stream=True,
)
text = "".join(c.choices[0].delta.content or "" for c in stream if c.choices)
assert text == "one two three"
def test_upstream_401_maps_to_authentication_error():
"""ClinePass answers a bad key with HTTP 401; that must not be flattened
into a generic APIConnectionError."""
from litellm.exceptions import AuthenticationError
def fake_post(self, url, *args, **kwargs):
raise httpx.HTTPStatusError(
"Unauthorized",
request=httpx.Request("POST", str(url)),
response=httpx.Response(
401,
json={"error": "Unauthorized"},
request=httpx.Request("POST", str(url)),
),
)
with patch.object(HTTPHandler, "post", fake_post):
with pytest.raises(AuthenticationError) as excinfo:
litellm.completion(
model="clinepass/deepseek-v4-flash",
messages=[{"role": "user", "content": "hi"}],
max_tokens=100,
)
assert excinfo.value.status_code == 401