diff --git a/litellm/__init__.py b/litellm/__init__.py index b6428b51bfb..40f90c4743a 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -2051,6 +2051,9 @@ if TYPE_CHECKING: ) from .llms.heroku.chat.transformation import HerokuChatConfig as HerokuChatConfig from .llms.cometapi.chat.transformation import CometAPIConfig as CometAPIConfig + from .llms.clinepass.chat.transformation import ( + ClinePassConfig as ClinePassConfig, + ) from .llms.azure.chat.gpt_transformation import ( AzureOpenAIConfig as AzureOpenAIConfig, ) diff --git a/litellm/_lazy_imports_registry.py b/litellm/_lazy_imports_registry.py index fcd2eed5387..aa2e8467946 100644 --- a/litellm/_lazy_imports_registry.py +++ b/litellm/_lazy_imports_registry.py @@ -284,6 +284,7 @@ LLM_CONFIG_NAMES: Final = ( "AzureOpenAIAssistantsAPIConfig", "HerokuChatConfig", "CometAPIConfig", + "ClinePassConfig", "AzureOpenAIConfig", "AzureOpenAIGPT5Config", "AzureOpenAITextConfig", @@ -1117,6 +1118,7 @@ _LLM_CONFIGS_IMPORT_MAP: Final = { ), "HerokuChatConfig": (".llms.heroku.chat.transformation", "HerokuChatConfig"), "CometAPIConfig": (".llms.cometapi.chat.transformation", "CometAPIConfig"), + "ClinePassConfig": (".llms.clinepass.chat.transformation", "ClinePassConfig"), "AzureOpenAIConfig": (".llms.azure.chat.gpt_transformation", "AzureOpenAIConfig"), "AzureOpenAIGPT5Config": ( ".llms.azure.chat.gpt_5_transformation", diff --git a/litellm/constants.py b/litellm/constants.py index 49514fc4d0e..6e5de432c51 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -782,6 +782,7 @@ LITELLM_CHAT_PROVIDERS: Final = [ "lemonade", "docker_model_runner", "amazon_nova", + "clinepass", ] # Resolving these providers runs an OAuth device flow (their provider info IS the login), so any @@ -1041,6 +1042,7 @@ openai_compatible_providers: Final[list] = [ "scx-ai", "prism", "sail", + "clinepass", # ClinePass (Cline API) - has its own module; listed here for exception mapping ] OPENAI_AUDIO_TRANSCRIPTION_PROVIDERS: Final = frozenset({"openai"} | frozenset(openai_compatible_providers)) diff --git a/litellm/litellm_core_utils/get_llm_provider_logic.py b/litellm/litellm_core_utils/get_llm_provider_logic.py index d4642ae2aad..18aeb55d5ec 100644 --- a/litellm/litellm_core_utils/get_llm_provider_logic.py +++ b/litellm/litellm_core_utils/get_llm_provider_logic.py @@ -611,6 +611,11 @@ def _get_openai_compatible_provider_info( api_base, dynamic_api_key, ) = litellm.PerplexityChatConfig()._get_openai_compatible_provider_info(api_base, api_key) + elif custom_llm_provider == "clinepass": + ( + api_base, + dynamic_api_key, + ) = litellm.ClinePassConfig()._get_openai_compatible_provider_info(api_base, api_key) elif custom_llm_provider == "aiohttp_openai": return model, "aiohttp_openai", api_key, api_base elif custom_llm_provider == "anyscale": diff --git a/litellm/llms/clinepass/chat/transformation.py b/litellm/llms/clinepass/chat/transformation.py new file mode 100644 index 00000000000..7b077467ebd --- /dev/null +++ b/litellm/llms/clinepass/chat/transformation.py @@ -0,0 +1,209 @@ +""" +Support for ClinePass (the Cline API) `/v1/chat/completions` endpoint. + +ClinePass is OpenAI-compatible apart from two quirks, both handled here: + +1. Non-streaming completions are nested under a ``data`` envelope -- + ``{"data": {"choices": [...]}, "success": true}`` -- rather than returning + ``choices`` at the top level. Streaming responses are *not* wrapped, so the + inherited SSE handling needs no change. +2. A bare model id is rejected with HTTP 400 ``invalid model format. Expected + format: modelType/model``, but LiteLLM strips its own ``clinepass/`` routing + prefix before the request is built, so it has to be restored. + +Documentation: https://docs.cline.bot/ +""" + +import json +from typing import Any, List, Tuple, Union + +import httpx + +from litellm.llms.base_llm.chat.transformation import BaseLLMException +from litellm.secret_managers.main import get_secret_str +from litellm.types.llms.openai import AllMessageValues +from litellm.types.utils import ModelResponse + +from ...openai.chat.gpt_transformation import OpenAIGPTConfig +from ..common_utils import ClinePassException + +CLINEPASS_API_BASE = "https://api.cline.bot/api/v1" + +# ClinePass nests the completion under this key on non-streaming responses. +CLINEPASS_RESPONSE_ENVELOPE_KEY = "data" + +# The qualifier ClinePass requires on outbound model ids. +CLINEPASS_MODEL_PREFIX = "clinepass/" + +# Headers that describe the original byte stream and would be wrong once the +# body is rewritten by _unwrap_response_envelope(). +_BODY_SPECIFIC_HEADERS = ("content-length", "content-encoding") + + +def _unwrap_response_envelope(raw_response: httpx.Response) -> httpx.Response: + """Strip ClinePass's ``data`` wrapper off a JSON completion body. + + The OpenAI transforms read ``raw_response.json()`` directly, so rebuild the + response around the inner object rather than duplicating their bodies here. + + Returns the original response untouched whenever the body does not look like + a wrapped completion, so an already-OpenAI-shaped body -- or an error nested + under the same key -- is not mistaken for one. + """ + try: + payload = raw_response.json() + except (ValueError, httpx.StreamError): + # Not a JSON body, or a streaming response that has not been read -- + # either way there is no envelope to strip. + return raw_response + + if not isinstance(payload, dict) or "choices" in payload: + return raw_response + + inner = payload.get(CLINEPASS_RESPONSE_ENVELOPE_KEY) + if not isinstance(inner, dict) or "choices" not in inner: + return raw_response + + headers = {k: v for k, v in raw_response.headers.items() if k.lower() not in _BODY_SPECIFIC_HEADERS} + + return httpx.Response( + status_code=raw_response.status_code, + headers=headers, + content=json.dumps(inner).encode("utf-8"), + request=getattr(raw_response, "_request", None), + ) + + +def _apply_model_prefix(data: dict) -> dict: + """Restore the ``modelType/model`` qualifier on the outbound model id. + + Only prefix ids that lost their qualifier, so a cross-provider id + (``clinepass/openrouter/foo`` -> ``openrouter/foo``) is forwarded unchanged. + """ + model = data.get("model") + if isinstance(model, str) and "/" not in model: + data["model"] = f"{CLINEPASS_MODEL_PREFIX}{model}" + return data + + +class ClinePassConfig(OpenAIGPTConfig): + """ + ClinePass configuration, inheriting the OpenAI chat transforms. + + Overrides only the request/response points where ClinePass diverges; see the + module docstring for the two quirks. + """ + + def _get_openai_compatible_provider_info( + self, api_base: str | None, api_key: str | None + ) -> Tuple[str | None, str | None]: + api_base = api_base or get_secret_str("CLINEPASS_API_BASE") or CLINEPASS_API_BASE + dynamic_api_key = api_key or get_secret_str("CLINEPASS_API_KEY") + return api_base, dynamic_api_key + + def get_complete_url( + self, + api_base: str | None, + api_key: str | None, + model: str, + optional_params: dict, + litellm_params: dict, + stream: bool | None = None, + ) -> str: + if not api_base: + api_base = CLINEPASS_API_BASE + + api_base = api_base.rstrip("/") + if api_base.endswith("/chat/completions"): + return api_base + + return f"{api_base}/chat/completions" + + def map_openai_params( + self, + non_default_params: dict, + optional_params: dict, + model: str, + drop_params: bool, + ) -> dict: + """ClinePass takes the legacy ``max_tokens`` spelling only.""" + mapped_params = super().map_openai_params( + non_default_params=non_default_params, + optional_params=optional_params, + model=model, + drop_params=drop_params, + ) + if "max_completion_tokens" in mapped_params: + mapped_params["max_tokens"] = mapped_params.pop("max_completion_tokens") + return mapped_params + + def transform_request( + self, + model: str, + messages: List[AllMessageValues], + optional_params: dict, + litellm_params: dict, + headers: dict, + ) -> dict: + data = super().transform_request( + model=model, + messages=messages, + optional_params=optional_params, + litellm_params=litellm_params, + headers=headers, + ) + return _apply_model_prefix(data) + + async def async_transform_request( + self, + model: str, + messages: List[AllMessageValues], + optional_params: dict, + litellm_params: dict, + headers: dict, + ) -> dict: + data = await super().async_transform_request( + model=model, + messages=messages, + optional_params=optional_params, + litellm_params=litellm_params, + headers=headers, + ) + return _apply_model_prefix(data) + + def transform_response( + self, + model: str, + raw_response: httpx.Response, + model_response: ModelResponse, + logging_obj: Any, + request_data: dict, + messages: List[AllMessageValues], + optional_params: dict, + litellm_params: dict, + encoding: Any, + api_key: str | None = None, + json_mode: bool | None = None, + ) -> ModelResponse: + return super().transform_response( + model=model, + raw_response=_unwrap_response_envelope(raw_response), + model_response=model_response, + logging_obj=logging_obj, + request_data=request_data, + messages=messages, + optional_params=optional_params, + litellm_params=litellm_params, + encoding=encoding, + api_key=api_key, + json_mode=json_mode, + ) + + def get_error_class( + self, error_message: str, status_code: int, headers: Union[dict, httpx.Headers] + ) -> BaseLLMException: + return ClinePassException( + message=error_message, + status_code=status_code, + headers=headers, + ) diff --git a/litellm/llms/clinepass/common_utils.py b/litellm/llms/clinepass/common_utils.py new file mode 100644 index 00000000000..27d6db446eb --- /dev/null +++ b/litellm/llms/clinepass/common_utils.py @@ -0,0 +1,7 @@ +from litellm.llms.base_llm.chat.transformation import BaseLLMException + + +class ClinePassException(BaseLLMException): + """ClinePass exception handling class""" + + pass diff --git a/litellm/main.py b/litellm/main.py index a818213b861..789f8ae6913 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -2448,6 +2448,57 @@ def _complete_aiohttp_openai( ) +def _complete_clinepass(ctx: _CompletionDispatchContext) -> _CompletionDispatchResult: + acompletion = ctx.acompletion + api_base = ctx.api_base + api_key = ctx.api_key + client = ctx.client + custom_llm_provider = ctx.custom_llm_provider + headers = ctx.headers + litellm_params = ctx.litellm_params + logging = ctx.logging + messages = ctx.messages + model = ctx.model + model_response = ctx.model_response + optional_params = ctx.optional_params + provider_config = ctx.provider_config + shared_session = ctx.shared_session + stream = ctx.stream + timeout = ctx.timeout + + api_key = api_key or get_secret_str("CLINEPASS_API_KEY") or litellm.api_key + + api_base = ( + api_base or litellm.api_base or get_secret_str("CLINEPASS_API_BASE") or "https://api.cline.bot/api/v1" + ) + + ## COMPLETION CALL + response = base_llm_http_handler.completion( + model=model, + messages=messages, + headers=headers, + model_response=model_response, + api_key=api_key, + api_base=api_base, + acompletion=acompletion, + logging_obj=logging, + optional_params=optional_params, + litellm_params=litellm_params, + shared_session=shared_session, + timeout=timeout, + client=client, + custom_llm_provider=custom_llm_provider, + encoding=_get_encoding(), + stream=stream, + provider_config=provider_config, + ) + + ## LOGGING + logging.post_call(input=messages, api_key=api_key, original_response=response) + + return response + + def _complete_cometapi(ctx: _CompletionDispatchContext) -> _CompletionDispatchResult: acompletion: Final = ctx.acompletion api_base = ctx.api_base @@ -5919,6 +5970,8 @@ def completion( response = _complete_aiohttp_openai(_dispatch_ctx) elif custom_llm_provider == "cometapi": response = _complete_cometapi(_dispatch_ctx) + elif custom_llm_provider == "clinepass": + response = _complete_clinepass(_dispatch_ctx) elif custom_llm_provider == "minimax": response = _complete_minimax(_dispatch_ctx) elif custom_llm_provider == "hosted_vllm": diff --git a/litellm/types/utils.py b/litellm/types/utils.py index a33eeaccaa3..754200cf5aa 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -4156,6 +4156,7 @@ class LlmProviders(str, Enum): APERTIS = "apertis" NANOGPT = "nano-gpt" POE = "poe" + CLINEPASS = "clinepass" CHUTES = "chutes" NEOSANTARA = "neosantara" PARASAIL = "parasail" diff --git a/litellm/utils.py b/litellm/utils.py index d72588e2b00..ab840f4016b 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -8537,6 +8537,7 @@ class ProviderConfigManager: LlmProviders.EDENAI: (litellm.EdenAIChatConfig, False), LlmProviders.FAL_AI: (litellm.FalAIChatConfig, False), LlmProviders.COMETAPI: (lambda: litellm.CometAPIConfig(), False), + LlmProviders.CLINEPASS: (lambda: litellm.ClinePassConfig(), False), LlmProviders.DATAROBOT: (lambda: litellm.DataRobotConfig(), False), LlmProviders.GEMINI: (lambda: litellm.GoogleAIStudioGeminiConfig(), False), LlmProviders.AI21: (lambda: litellm.AI21ChatConfig(), False), diff --git a/tests/test_litellm/llms/clinepass/chat/test_clinepass_transformation.py b/tests/test_litellm/llms/clinepass/chat/test_clinepass_transformation.py new file mode 100644 index 00000000000..e29968dcda0 --- /dev/null +++ b/tests/test_litellm/llms/clinepass/chat/test_clinepass_transformation.py @@ -0,0 +1,295 @@ +"""Tests for the ClinePass provider. + +The point of the end-to-end tests here is that they drive ``litellm.completion()`` +with a mocked transport rather than calling the transforms directly -- a unit test +that calls ``transform_response()`` itself proves the function is correct but not +that anything invokes it, which is exactly how the envelope unwrap was previously +shipped as dead code. +""" + +import json +from unittest.mock import patch + +import httpx +import pytest + +import litellm +from litellm.llms.clinepass.chat.transformation import ( + ClinePassConfig, + _apply_model_prefix, + _unwrap_response_envelope, +) +from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler +from litellm.types.utils import LlmProviders +from litellm.utils import ProviderConfigManager + +API_KEY = "sk-clinepass-test-not-real" + +ENVELOPED_COMPLETION = { + "success": True, + "data": { + "id": "chatcmpl-test", + "object": "chat.completion", + "created": 1, + "model": "clinepass/deepseek-v4-flash", + "choices": [ + { + "index": 0, + "finish_reason": "stop", + "message": { + "role": "assistant", + "content": "pong", + "reasoning": "the user asked for pong", + }, + } + ], + "usage": {"prompt_tokens": 5, "completion_tokens": 2, "total_tokens": 7}, + }, +} + + +def _response(payload: dict, url: str = "https://api.cline.bot/api/v1/chat/completions") -> httpx.Response: + return httpx.Response(200, json=payload, request=httpx.Request("POST", url)) + + +@pytest.fixture(autouse=True) +def _clinepass_env(monkeypatch): + monkeypatch.setenv("CLINEPASS_API_KEY", API_KEY) + monkeypatch.delenv("CLINEPASS_API_BASE", raising=False) + + +# -------------------------------------------------------------------------- +# Registration / routing +# -------------------------------------------------------------------------- + + +def test_get_llm_provider_resolves_clinepass(): + model, provider, api_key, api_base = litellm.get_llm_provider(model="clinepass/deepseek-v4-flash") + assert model == "deepseek-v4-flash" + assert provider == "clinepass" + assert api_key == API_KEY + assert api_base == "https://api.cline.bot/api/v1" + + +def test_provider_config_manager_returns_clinepass_config(): + config = ProviderConfigManager.get_provider_chat_config(model="deepseek-v4-flash", provider=LlmProviders.CLINEPASS) + assert isinstance(config, ClinePassConfig) + + +def test_clinepass_is_not_a_json_configured_provider(): + """ClinePass needs a response transform, which the JSON provider system's + OpenAI-SDK dispatch path never invokes. Guard against it drifting back.""" + from litellm.llms.openai_like.json_loader import JSONProviderRegistry + + assert not JSONProviderRegistry.exists("clinepass") + + +def test_clinepass_stays_in_openai_compatible_providers(): + """Membership drives `_map_openai_exception`, so dropping it silently + downgrades a 401 to APIConnectionError. The explicit dispatch branch in + main.py precedes the openai_compatible_providers catch-all, so being listed + here does NOT route ClinePass to the OpenAI SDK path.""" + assert "clinepass" in litellm.openai_compatible_providers + + +def test_api_base_env_override(monkeypatch): + monkeypatch.setenv("CLINEPASS_API_BASE", "https://proxy.internal/api/v1") + _, _, _, api_base = litellm.get_llm_provider(model="clinepass/deepseek-v4-flash") + assert api_base == "https://proxy.internal/api/v1" + + +@pytest.mark.parametrize( + "api_base,expected", + [ + (None, "https://api.cline.bot/api/v1/chat/completions"), + ("https://api.cline.bot/api/v1", "https://api.cline.bot/api/v1/chat/completions"), + ("https://api.cline.bot/api/v1/", "https://api.cline.bot/api/v1/chat/completions"), + ( + "https://api.cline.bot/api/v1/chat/completions", + "https://api.cline.bot/api/v1/chat/completions", + ), + ], +) +def test_get_complete_url(api_base, expected): + url = ClinePassConfig().get_complete_url( + api_base=api_base, api_key=API_KEY, model="deepseek-v4-flash", optional_params={}, litellm_params={} + ) + assert url == expected + + +# -------------------------------------------------------------------------- +# Model prefix +# -------------------------------------------------------------------------- + + +def test_model_prefix_restored_on_bare_id(): + assert _apply_model_prefix({"model": "deepseek-v4-flash"})["model"] == "clinepass/deepseek-v4-flash" + + +def test_model_prefix_left_alone_when_qualifier_present(): + """`clinepass/openrouter/foo` arrives here as `openrouter/foo` and must pass through.""" + assert _apply_model_prefix({"model": "openrouter/foo"})["model"] == "openrouter/foo" + + +def test_model_prefix_ignores_missing_model(): + assert _apply_model_prefix({}) == {} + + +# -------------------------------------------------------------------------- +# Response envelope +# -------------------------------------------------------------------------- + + +def test_unwrap_envelope_extracts_inner_completion(): + unwrapped = _unwrap_response_envelope(_response(ENVELOPED_COMPLETION)) + assert unwrapped.json() == ENVELOPED_COMPLETION["data"] + + +def test_unwrap_envelope_content_length_describes_the_new_body(): + """The original content-length describes the enveloped bytes and must not be + carried over; httpx recomputes a correct one for the rewritten body.""" + raw = _response(ENVELOPED_COMPLETION) + unwrapped = _unwrap_response_envelope(raw) + assert unwrapped.headers["content-length"] != raw.headers["content-length"] + assert int(unwrapped.headers["content-length"]) == len(unwrapped.content) + + +def test_unwrap_envelope_passes_through_openai_shaped_body(): + payload = ENVELOPED_COMPLETION["data"] + assert _unwrap_response_envelope(_response(payload)).json() == payload + + +def test_unwrap_envelope_passes_through_error_nested_under_same_key(): + """An error under `data` has no `choices` and must not be mistaken for a completion.""" + payload = {"success": False, "data": {"message": "bad model"}} + assert _unwrap_response_envelope(_response(payload)).json() == payload + + +def test_unwrap_envelope_passes_through_non_json_body(): + raw = httpx.Response( + 200, content=b"not json", request=httpx.Request("POST", "https://api.cline.bot/api/v1/chat/completions") + ) + assert _unwrap_response_envelope(raw) is raw + + +# -------------------------------------------------------------------------- +# Parameter mapping +# -------------------------------------------------------------------------- + + +def test_max_completion_tokens_mapped_to_max_tokens(): + mapped = ClinePassConfig().map_openai_params( + non_default_params={"max_completion_tokens": 4000}, + optional_params={}, + model="deepseek-v4-flash", + drop_params=False, + ) + assert mapped == {"max_tokens": 4000} + + +# -------------------------------------------------------------------------- +# End-to-end through litellm.completion() -- these are the load-bearing ones +# -------------------------------------------------------------------------- + + +def test_completion_unwraps_envelope_and_prefixes_model(): + captured = {} + + def fake_post(self, url, *args, **kwargs): + captured["url"] = str(url) + captured["body"] = json.loads(kwargs["data"]) + return _response(ENVELOPED_COMPLETION) + + with patch.object(HTTPHandler, "post", fake_post): + response = litellm.completion( + model="clinepass/deepseek-v4-flash", + messages=[{"role": "user", "content": "ping"}], + max_tokens=4000, + ) + + assert captured["url"] == "https://api.cline.bot/api/v1/chat/completions" + assert captured["body"]["model"] == "clinepass/deepseek-v4-flash" + assert response.choices[0].message.content == "pong" + assert response.choices[0].message.reasoning_content == "the user asked for pong" + + +@pytest.mark.asyncio +async def test_acompletion_unwraps_envelope_and_prefixes_model(): + captured = {} + + async def fake_post(self, url, *args, **kwargs): + captured["url"] = str(url) + captured["body"] = json.loads(kwargs["data"]) + return _response(ENVELOPED_COMPLETION) + + with patch.object(AsyncHTTPHandler, "post", fake_post): + response = await litellm.acompletion( + model="clinepass/deepseek-v4-flash", + messages=[{"role": "user", "content": "ping"}], + max_tokens=4000, + ) + + assert captured["url"] == "https://api.cline.bot/api/v1/chat/completions" + assert captured["body"]["model"] == "clinepass/deepseek-v4-flash" + assert response.choices[0].message.content == "pong" + + +def test_completion_streaming_is_not_unwrapped(): + """ClinePass does NOT wrap SSE chunks -- they are already OpenAI-shaped.""" + chunks = [ + { + "id": "chatcmpl-test", + "object": "chat.completion.chunk", + "created": 1, + "model": "clinepass/deepseek-v4-flash", + "choices": [{"index": 0, "delta": {"content": piece}, "finish_reason": None}], + } + for piece in ["one ", "two ", "three"] + ] + body = "".join(f"data: {json.dumps(c)}\n\n" for c in chunks) + "data: [DONE]\n\n" + + def fake_post(self, url, *args, **kwargs): + return httpx.Response( + 200, + content=body.encode(), + headers={"content-type": "text/event-stream"}, + request=httpx.Request("POST", str(url)), + ) + + with patch.object(HTTPHandler, "post", fake_post): + stream = litellm.completion( + model="clinepass/deepseek-v4-flash", + messages=[{"role": "user", "content": "count"}], + max_tokens=4000, + stream=True, + ) + text = "".join(c.choices[0].delta.content or "" for c in stream if c.choices) + + assert text == "one two three" + + +def test_upstream_401_maps_to_authentication_error(): + """ClinePass answers a bad key with HTTP 401; that must not be flattened + into a generic APIConnectionError.""" + from litellm.exceptions import AuthenticationError + + def fake_post(self, url, *args, **kwargs): + raise httpx.HTTPStatusError( + "Unauthorized", + request=httpx.Request("POST", str(url)), + response=httpx.Response( + 401, + json={"error": "Unauthorized"}, + request=httpx.Request("POST", str(url)), + ), + ) + + with patch.object(HTTPHandler, "post", fake_post): + with pytest.raises(AuthenticationError) as excinfo: + litellm.completion( + model="clinepass/deepseek-v4-flash", + messages=[{"role": "user", "content": "hi"}], + max_tokens=100, + ) + + assert excinfo.value.status_code == 401