This commit is contained in:
Daniel JB Clark 2026-10-04 16:39:23 -07:00 • committed by GitHub
commit cfa54fdb7e
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
25 changed files with 1626 additions and 10 deletions

View file

@ -317,6 +317,7 @@ Set `LITELLM_PROXY_API_BASE` and `LITELLM_PROXY_API_KEY` and every model call th
| [Bytez (`bytez`)](https://docs.litellm.ai/docs/providers/bytez) | ✅ | ✅ | ✅ | | | | | | | |
| [Cerebras (`cerebras`)](https://docs.litellm.ai/docs/providers/cerebras) | ✅ | ✅ | ✅ | | | | | | | |
| [Clarifai (`clarifai`)](https://docs.litellm.ai/docs/providers/clarifai) | ✅ | ✅ | ✅ | | | | | | | |
| [ClinePass (`clinepass`)](https://docs.litellm.ai/docs/providers/clinepass) | ✅ | ✅ | ✅ | | | | | | | |
| [Cloudflare AI Workers (`cloudflare`)](https://docs.litellm.ai/docs/providers/cloudflare_workers) | ✅ | ✅ | ✅ | | | | | | | |
| [Codestral (`codestral`)](https://docs.litellm.ai/docs/providers/codestral) | ✅ | ✅ | ✅ | | | | | | | |
| [Cognition (`cognition`)](https://docs.litellm.ai/docs/providers/cognition) | ✅ | ✅ | ✅ | | | | | | | |

View file

@ -644,6 +644,7 @@ gemini_models: Set = set()
xai_models: Set = set()
zai_models: Set = set()
deepseek_models: Set = set()
clinepass_models: Set = set()
tencent_models: Set = set()
runwayml_models: Set = set()
azure_ai_models: Set = set()
@ -866,6 +867,8 @@ def _populate_provider_model_sets(model_cost_map: Dict) -> None:
fal_ai_models.add(key)
elif value.get("litellm_provider") == "deepseek":
deepseek_models.add(key)
elif value.get("litellm_provider") == "clinepass":
clinepass_models.add(key) # pyright: ignore[reportUnknownArgumentType] # key comes from the untyped model cost map
elif value.get("litellm_provider") == "tencent":
tencent_models.add(key)
elif value.get("litellm_provider") == "runwayml":
@ -1078,6 +1081,7 @@ model_list = list(
| zai_models
| fal_ai_models
| deepseek_models
| clinepass_models
| modelscope_models
| azure_ai_models
| voyage_models
@ -1183,6 +1187,7 @@ def _build_models_by_provider() -> dict:
"zai": zai_models,
"fal_ai": fal_ai_models,
"deepseek": deepseek_models,
"clinepass": clinepass_models,
"tencent": tencent_models,
"runwayml": runwayml_models,
"mistral": mistral_chat_models,
@ -2052,6 +2057,9 @@ if TYPE_CHECKING:
)
from .llms.heroku.chat.transformation import HerokuChatConfig as HerokuChatConfig
from .llms.cometapi.chat.transformation import CometAPIConfig as CometAPIConfig
from .llms.clinepass.chat.transformation import (
ClinePassConfig as ClinePassConfig,
)
from .llms.azure.chat.gpt_transformation import (
AzureOpenAIConfig as AzureOpenAIConfig,
)

View file

@ -284,6 +284,7 @@ LLM_CONFIG_NAMES: Final = (
"AzureOpenAIAssistantsAPIConfig",
"HerokuChatConfig",
"CometAPIConfig",
"ClinePassConfig",
"AzureOpenAIConfig",
"AzureOpenAIGPT5Config",
"AzureOpenAITextConfig",
@ -1117,6 +1118,7 @@ _LLM_CONFIGS_IMPORT_MAP: Final = {
),
"HerokuChatConfig": (".llms.heroku.chat.transformation", "HerokuChatConfig"),
"CometAPIConfig": (".llms.cometapi.chat.transformation", "CometAPIConfig"),
"ClinePassConfig": (".llms.clinepass.chat.transformation", "ClinePassConfig"),
"AzureOpenAIConfig": (".llms.azure.chat.gpt_transformation", "AzureOpenAIConfig"),
"AzureOpenAIGPT5Config": (
".llms.azure.chat.gpt_5_transformation",

View file

@ -783,6 +783,7 @@ LITELLM_CHAT_PROVIDERS: Final = [
"lemonade",
"docker_model_runner",
"amazon_nova",
"clinepass",
]
# Resolving these providers runs an OAuth device flow (their provider info IS the login), so any

View file

@ -2468,6 +2468,7 @@ def exception_type(
or custom_llm_provider == "text-completion-openai"
or custom_llm_provider == "custom_openai"
or custom_llm_provider in litellm.openai_compatible_providers
or custom_llm_provider == "clinepass"
or custom_llm_provider == "mistral"
or custom_llm_provider == "runwayml"
):

View file

@ -611,6 +611,11 @@ def _get_openai_compatible_provider_info(
api_base,
dynamic_api_key,
) = litellm.PerplexityChatConfig()._get_openai_compatible_provider_info(api_base, api_key)
elif custom_llm_provider == "clinepass":
(
api_base,
dynamic_api_key,
) = litellm.ClinePassConfig()._get_openai_compatible_provider_info(api_base, api_key)
elif custom_llm_provider == "aiohttp_openai":
return model, "aiohttp_openai", api_key, api_base
elif custom_llm_provider == "anyscale":

View file

@ -0,0 +1,268 @@
"""
Support for ClinePass (the Cline API) `/v1/chat/completions` endpoint.
ClinePass is OpenAI-compatible apart from two quirks, both handled here:
1. Non-streaming completions are nested under a ``data`` envelope --
``{"data": {"choices": [...]}, "success": true}`` -- rather than returning
``choices`` at the top level. Streaming responses are *not* wrapped, so the
inherited SSE handling needs no change.
2. A bare model id is rejected with HTTP 400 ``invalid model format. Expected
format: modelType/model``, but LiteLLM strips its own ``clinepass/`` routing
prefix before the request is built, so a qualifier has to be restored.
Documentation: https://docs.cline.bot/
Credentials come only from the request's api_key or CLINEPASS_API_KEY.
Moderation and realtime endpoints are unsupported and rejected before dispatch.
"""
import json
from typing import TYPE_CHECKING, Any, Final, NoReturn
import httpx
from litellm.llms.base_llm.chat.transformation import BaseLLMException
from litellm.secret_managers.main import get_secret_str
from litellm.types.llms.openai import AllMessageValues
from litellm.types.utils import ModelResponse
from ...openai.chat.gpt_transformation import OpenAIGPTConfig
from ..common_utils import ClinePassException
# Mirrors litellm/llms/openai/chat/gpt_transformation.py: these are needed only
# for annotations, and importing litellm_logging at runtime from a provider
# module risks a circular import.
if TYPE_CHECKING:
from litellm.litellm_core_utils.litellm_logging import Logging as _LiteLLMLoggingObj
from litellm.litellm_core_utils.tokenizer import Encoding as Tokenizer
LiteLLMLoggingObj = _LiteLLMLoggingObj
else:
LiteLLMLoggingObj = Any
CLINEPASS_API_BASE: Final = "https://api.cline.bot/api/v1"
# ClinePass nests the completion under this key on non-streaming responses.
CLINEPASS_RESPONSE_ENVELOPE_KEY: Final = "data"
# The qualifier ClinePass expects on outbound model ids.
#
# Note the hyphen: the catalog namespace is ``cline-pass/``, not ``clinepass/``
# (the latter is LiteLLM's own routing prefix, which is stripped before the
# request is built). The API only validates the *shape* of a model id -- any
# ``<segment>/<model>`` is accepted with HTTP 200 -- so an incorrect namespace
# fails silently rather than loudly. It is not inert, though: for at least one
# model an unrecognised namespace resolves to a different, date-pinned snapshot
# (``cline-pass/deepseek-v4-flash`` -> ``deepseek/deepseek-v4-flash``, while
# ``clinepass/deepseek-v4-flash`` -> ``deepseek/deepseek-v4-flash-0731``).
CLINEPASS_MODEL_PREFIX: Final = "cline-pass/"
# Headers that describe the original byte stream and would be wrong once the
# body is rewritten by _unwrap_response_envelope().
_BODY_SPECIFIC_HEADERS: Final = ("content-length", "content-encoding")
def _unwrap_response_envelope(raw_response: httpx.Response) -> httpx.Response:
"""Strip ClinePass's ``data`` wrapper off a JSON completion body.
The OpenAI transforms read ``raw_response.json()`` directly, so rebuild the
response around the inner object rather than duplicating their bodies here.
Returns the original response untouched whenever the body does not look like
a wrapped completion, so an already-OpenAI-shaped body -- or an error nested
under the same key -- is not mistaken for one.
"""
try:
payload = raw_response.json()
except ValueError:
# Not a JSON body -- there is no envelope to strip.
return raw_response
if not isinstance(payload, dict) or "choices" in payload:
return raw_response
inner = payload.get(CLINEPASS_RESPONSE_ENVELOPE_KEY)
if not isinstance(inner, dict) or "choices" not in inner:
return raw_response
headers = {k: v for k, v in raw_response.headers.items() if k.lower() not in _BODY_SPECIFIC_HEADERS}
# httpx.Response.request raises RuntimeError rather than returning None when
# no request is attached, so ask for it defensively instead of reaching for
# the private attribute behind it.
try:
original_request = raw_response.request
except RuntimeError:
original_request = None
return httpx.Response(
status_code=raw_response.status_code,
headers=headers,
content=json.dumps(inner, ensure_ascii=False).encode("utf-8"),
request=original_request,
)
def _apply_model_prefix(data: dict) -> dict: # mutable-ok: request body handed to the dict-typed base transform_request
"""Restore the ``modelType/model`` qualifier on the outbound model id.
Only prefix ids that lost their qualifier, so a cross-provider id
(``clinepass/openrouter/foo`` -> ``openrouter/foo``) is forwarded unchanged.
"""
model = data.get("model")
if isinstance(model, str) and "/" not in model:
data["model"] = f"{CLINEPASS_MODEL_PREFIX}{model}"
return data
class ClinePassConfig(OpenAIGPTConfig):
"""
ClinePass configuration, inheriting the OpenAI chat transforms.
Overrides only the request/response points where ClinePass diverges; see the
module docstring for the two quirks.
"""
@staticmethod
def get_realtime_http_config(model: str) -> NoReturn:
from litellm.exceptions import BadRequestError
raise BadRequestError(
message="ClinePass does not support realtime endpoints",
model=model,
llm_provider="clinepass",
)
@staticmethod
def validate_moderation(model: str | None, custom_llm_provider: str | None = None) -> None:
if custom_llm_provider != "clinepass" and not (model or "").startswith("clinepass/"):
return
from litellm.exceptions import BadRequestError
raise BadRequestError(
message="ClinePass does not support moderation endpoints",
model=model or "",
llm_provider="clinepass",
)
def _get_openai_compatible_provider_info(
self, api_base: str | None, api_key: str | None
) -> tuple[str | None, str | None]:
api_base = api_base or get_secret_str("CLINEPASS_API_BASE") or CLINEPASS_API_BASE
dynamic_api_key = api_key or get_secret_str("CLINEPASS_API_KEY")
return api_base, dynamic_api_key
def get_complete_url(
self,
api_base: str | None,
api_key: str | None,
model: str,
optional_params: dict, # mutable-ok: matches the dict-typed base-class signature
litellm_params: dict, # mutable-ok: matches the dict-typed base-class signature
stream: bool | None = None,
) -> str:
if not api_base:
api_base = CLINEPASS_API_BASE
api_base = api_base.rstrip("/")
if api_base.endswith("/chat/completions"):
return api_base
return f"{api_base}/chat/completions"
def get_models(
self, api_key: str | None = None, api_base: str | None = None
) -> list[str]: # mutable-ok: matches the dict-typed base-class signature
"""ClinePass exposes no model catalog.
``GET https://api.cline.bot/api/v1/models`` returns HTTP 404, and the
inherited OpenAI implementation would additionally ask for it at the
wrong path -- it rewrites the base URL down to scheme+host and appends
``/v1/models``. Return an empty catalog rather than making a request
that is known to fail.
"""
return []
def map_openai_params(
self,
non_default_params: dict, # mutable-ok: matches the dict-typed base-class signature
optional_params: dict, # mutable-ok: matches the dict-typed base-class signature
model: str,
drop_params: bool,
) -> dict: # mutable-ok: matches the dict-typed base-class signature
"""ClinePass takes the legacy ``max_tokens`` spelling only."""
mapped_params = super().map_openai_params(
non_default_params=non_default_params,
optional_params=optional_params,
model=model,
drop_params=drop_params,
)
if "max_completion_tokens" in mapped_params:
mapped_params["max_tokens"] = mapped_params.pop("max_completion_tokens")
return mapped_params
def transform_request(
self,
model: str,
messages: list[AllMessageValues], # mutable-ok: matches the dict-typed base-class signature
optional_params: dict, # mutable-ok: matches the dict-typed base-class signature
litellm_params: dict, # mutable-ok: matches the dict-typed base-class signature
headers: dict, # mutable-ok: matches the dict-typed base-class signature
) -> dict: # mutable-ok: matches the dict-typed base-class signature
# BaseLLMHTTPHandler builds the body with this synchronous method on
# both the sync and the async path, so there is deliberately no
# async_transform_request() override -- it would never be called.
data = super().transform_request(
model=model,
messages=messages,
optional_params=optional_params,
litellm_params=litellm_params,
headers=headers,
)
return _apply_model_prefix(data)
def transform_response(
self,
model: str,
raw_response: httpx.Response,
model_response: ModelResponse,
logging_obj: LiteLLMLoggingObj,
request_data: dict, # mutable-ok: matches the dict-typed base-class signature
messages: list[AllMessageValues], # mutable-ok: matches the dict-typed base-class signature
optional_params: dict, # mutable-ok: matches the dict-typed base-class signature
litellm_params: dict, # mutable-ok: matches the dict-typed base-class signature
encoding: "Tokenizer | None",
api_key: str | None = None,
json_mode: bool | None = None,
) -> ModelResponse:
# ClinePass was once observed returning finish_reason "stop" on a completion
# cut off by max_tokens. Follow-up probes on 2026-08-22 did not reproduce it.
# The provider therefore reports the upstream finish reason unmodified: inferring
# truncation from usage equalling the cap produces false positives on natural
# completions that happen to land exactly on the cap.
return super().transform_response(
model=model,
raw_response=_unwrap_response_envelope(raw_response),
model_response=model_response,
logging_obj=logging_obj,
request_data=request_data,
messages=messages,
optional_params=optional_params,
litellm_params=litellm_params,
encoding=encoding,
api_key=api_key,
json_mode=json_mode,
)
def get_error_class(
self,
error_message: str,
status_code: int,
headers: dict | httpx.Headers, # mutable-ok: matches the dict-typed base-class signature
) -> BaseLLMException:
return ClinePassException(
message=error_message,
status_code=status_code,
headers=headers,
)

View file

@ -0,0 +1,5 @@
from litellm.llms.base_llm.chat.transformation import BaseLLMException
class ClinePassException(BaseLLMException):
"""ClinePass exception handling class"""

View file

@ -2450,6 +2450,47 @@ def _complete_aiohttp_openai(
)
def _complete_http_provider(ctx: _CompletionDispatchContext) -> _CompletionDispatchResult:
acompletion: Final = ctx.acompletion
api_base: Final = ctx.api_base
api_key: Final = ctx.api_key
client: Final = _dispatch_client_http(ctx)
custom_llm_provider: Final = ctx.custom_llm_provider
headers: Final = ctx.headers
litellm_params: Final = ctx.litellm_params
logging: Final = ctx.logging
messages: Final = ctx.messages
model: Final = ctx.model
model_response: Final = ctx.model_response
optional_params: Final = ctx.optional_params
provider_config: Final = ctx.provider_config
shared_session: Final = ctx.shared_session
stream: Final = ctx.stream
timeout: Final = ctx.timeout
response: Final = base_llm_http_handler.completion(
model=model,
messages=messages, # pyright: ignore[reportUnknownArgumentType] # ctx.messages is list[Unknown]
headers=headers, # pyright: ignore[reportUnknownArgumentType] # ctx.headers is dict[Unknown, Unknown]
model_response=model_response,
api_key=api_key,
api_base=api_base,
acompletion=acompletion,
logging_obj=logging,
optional_params=optional_params,
litellm_params=litellm_params,
shared_session=shared_session,
timeout=timeout,
client=client,
custom_llm_provider=custom_llm_provider,
encoding=_get_encoding(),
stream=stream,
provider_config=provider_config,
)
return response
def _complete_cometapi(ctx: _CompletionDispatchContext) -> _CompletionDispatchResult:
acompletion: Final = ctx.acompletion
api_base = ctx.api_base
@ -5928,6 +5969,8 @@ def completion(
response = _complete_aiohttp_openai(_dispatch_ctx)
elif custom_llm_provider == "cometapi":
response = _complete_cometapi(_dispatch_ctx)
elif custom_llm_provider == "clinepass":
response = _complete_http_provider(_dispatch_ctx)
elif custom_llm_provider == "minimax":
response = _complete_minimax(_dispatch_ctx)
elif custom_llm_provider == "hosted_vllm":
@ -7775,6 +7818,10 @@ def adapter_completion(*, adapter_id: str, **kwargs) -> BaseModel | AdapterCompl
def moderation(input: str, model: str | None = None, api_key: str | None = None, **kwargs) -> OpenAIModerationResponse:
custom_llm_provider: Final[object] = kwargs.get("custom_llm_provider")
litellm.ClinePassConfig.validate_moderation(
model=model, custom_llm_provider=custom_llm_provider if isinstance(custom_llm_provider, str) else None
)
# only supports open ai for now
api_key = api_key or litellm.api_key or litellm.openai_key or get_secret_str("OPENAI_API_KEY")
@ -7809,6 +7856,7 @@ async def amoderation(
) -> OpenAIModerationResponse:
from openai import AsyncOpenAI
litellm.ClinePassConfig.validate_moderation(model=model, custom_llm_provider=custom_llm_provider)
# only supports open ai for now
api_key = api_key or litellm.api_key or litellm.openai_key or get_secret_str("OPENAI_API_KEY")
optional_params: Final = GenericLiteLLMParams.model_validate(kwargs)

View file

@ -15614,6 +15614,10 @@
"prompt_cache_min_tokens": 1024,
"source": "https://platform.claude.com/docs/en/about-claude/pricing"
},
"clinepass/deepseek-v4-flash": {
"litellm_provider": "clinepass",
"mode": "chat"
},
"cloudflare/clef": {
"input_cost_per_token": 2.4e-07,
"litellm_provider": "cloudflare",

View file

@ -492,6 +492,24 @@
"interactions": true
}
},
"clinepass": {
"display_name": "ClinePass (`clinepass`)",
"url": "https://docs.litellm.ai/docs/providers/clinepass",
"endpoints": {
"chat_completions": true,
"messages": true,
"responses": true,
"embeddings": false,
"image_generations": false,
"audio_transcriptions": false,
"audio_speech": false,
"moderations": false,
"batches": false,
"rerank": false,
"a2a": true,
"interactions": true
}
},
"cloudflare": {
"display_name": "Cloudflare AI Workers (`cloudflare`)",
"url": "https://docs.litellm.ai/docs/providers/cloudflare_workers",

View file

@ -829,6 +829,34 @@
],
"default_model_placeholder": "gpt-3.5-turbo"
},
{
"provider": "CLINEPASS",
"provider_display_name": "ClinePass",
"litellm_provider": "clinepass",
"credential_fields": [
{
"key": "api_base",
"label": "API Base",
"placeholder": null,
"tooltip": null,
"required": false,
"field_type": "text",
"options": null,
"default_value": null
},
{
"key": "api_key",
"label": "API Key",
"placeholder": null,
"tooltip": null,
"required": false,
"field_type": "password",
"options": null,
"default_value": null
}
],
"default_model_placeholder": "gpt-3.5-turbo"
},
{
"provider": "CLOUDFLARE",
"provider_display_name": "Cloudflare",

View file

@ -2408,10 +2408,13 @@ async def _aresponses_websocket(
resolved_api_key: Final = (
dynamic_api_key
or api_key
or litellm_params.api_key
or litellm.api_key
or litellm.openai_key
or get_secret_str("OPENAI_API_KEY")
or (
(litellm.api_key or litellm.openai_key or get_secret_str("OPENAI_API_KEY"))
if responses_api_provider_config is not None
else None
)
)
# Extract params that we're passing explicitly to avoid duplicates in **kwargs

View file

@ -2824,9 +2824,10 @@ class ManagedResponsesWebSocketHandler:
def _inject_credentials(self, call_kwargs: dict[str, object], model: str | None = None) -> None:
"""Inject connection-level credentials and metadata into call_kwargs."""
if self.api_key is not None:
same_provider: Final = self._same_provider(model)
if self.api_key is not None and same_provider:
call_kwargs["api_key"] = self.api_key
if self.api_base is not None:
if self.api_base is not None and same_provider:
call_kwargs["api_base"] = self.api_base
if self.timeout is not None:
call_kwargs["timeout"] = self.timeout
@ -2835,7 +2836,7 @@ class ManagedResponsesWebSocketHandler:
# (e.g., connection is vertex_ai but event says openai/gpt-4), let litellm
# re-resolve from the model string. Same-provider model variants (e.g.,
# vertex_ai/gemini-2.0 -> vertex_ai/gemini-1.5) still inherit the provider.
if self.custom_llm_provider is not None and self._same_provider(model):
if self.custom_llm_provider is not None and same_provider:
call_kwargs["custom_llm_provider"] = self.custom_llm_provider
if self.litellm_metadata:
call_kwargs["litellm_metadata"] = dict(self.litellm_metadata)
@ -2960,11 +2961,18 @@ class ManagedResponsesWebSocketHandler:
call_kwargs: Final = self._build_base_call_kwargs(msg_obj)
call_kwargs["stream"] = True
# A frame that repeats the connection's public alias (model_group) must
# reuse the router-resolved self.model; passing the alias raw to
# litellm.aresponses fails in get_llm_provider. A genuinely different
# provider-prefixed per-frame model is still honored.
requested_model: Final[str | None] = _optional_str(call_kwargs.pop("model", None))
authorized_models: Final = (self.model, self.model_group, f"{self.custom_llm_provider}/{self.model}")
if (
self.user_api_key_dict is not None
and requested_model is not None
and requested_model not in authorized_models
):
await self._send_error(
"Changing models requires a new authorized WebSocket connection",
error_type="invalid_request_error",
)
return
model: Final[str] = (
self.model if requested_model is None or requested_model == self.model_group else requested_model
)

View file

@ -4166,6 +4166,7 @@ class LlmProviders(str, Enum):
APERTIS = "apertis"
NANOGPT = "nano-gpt"
POE = "poe"
CLINEPASS = "clinepass"
CHUTES = "chutes"
NEOSANTARA = "neosantara"
PARASAIL = "parasail"

View file

@ -8504,6 +8504,7 @@ class ProviderConfigManager:
LlmProviders.EDENAI: (litellm.EdenAIChatConfig, False),
LlmProviders.FAL_AI: (litellm.FalAIChatConfig, False),
LlmProviders.COMETAPI: (lambda: litellm.CometAPIConfig(), False),
LlmProviders.CLINEPASS: (litellm.ClinePassConfig, False),
LlmProviders.DATAROBOT: (lambda: litellm.DataRobotConfig(), False),
LlmProviders.GEMINI: (lambda: litellm.GoogleAIStudioGeminiConfig(), False),
LlmProviders.AI21: (lambda: litellm.AI21ChatConfig(), False),
@ -9716,6 +9717,8 @@ class ProviderConfigManager:
(POST /realtime/client_secrets and POST /realtime/calls).
"""
if LlmProviders.CLINEPASS == provider:
return litellm.ClinePassConfig.get_realtime_http_config(model=model)
if LlmProviders.OPENAI == provider:
from litellm.llms.openai.realtime.http_transformation import (
OpenAIRealtimeHTTPConfig,

View file

@ -15614,6 +15614,10 @@
"prompt_cache_min_tokens": 1024,
"source": "https://platform.claude.com/docs/en/about-claude/pricing"
},
"clinepass/deepseek-v4-flash": {
"litellm_provider": "clinepass",
"mode": "chat"
},
"cloudflare/clef": {
"input_cost_per_token": 2.4e-07,
"litellm_provider": "cloudflare",

View file

@ -546,6 +546,24 @@
"interactions": true
}
},
"clinepass": {
"display_name": "ClinePass (`clinepass`)",
"url": "https://docs.litellm.ai/docs/providers/clinepass",
"endpoints": {
"chat_completions": true,
"messages": true,
"responses": true,
"embeddings": false,
"image_generations": false,
"audio_transcriptions": false,
"audio_speech": false,
"moderations": false,
"batches": false,
"rerank": false,
"a2a": true,
"interactions": true
}
},
"cloudflare": {
"display_name": "Cloudflare AI Workers (`cloudflare`)",
"url": "https://docs.litellm.ai/docs/providers/cloudflare_workers",

View file

@ -812,6 +812,7 @@ PROVIDERS_WITH_A_HANDLER = (
"azure",
"azure_ai",
"bedrock",
"clinepass",
"cloudflare",
"cohere",
"databricks",
@ -1048,6 +1049,7 @@ PROVIDERS_THAT_RECOGNISE_A_FULL_CONTEXT_WINDOW = (
"anthropic",
"azure",
"azure_ai",
"clinepass",
"databricks",
"deepseek",
"fireworks_ai",
@ -1066,6 +1068,7 @@ PROVIDERS_THAT_RECOGNISE_A_CONTENT_POLICY_BLOCK = (
"ai21",
"azure",
"azure_ai",
"clinepass",
"deepseek",
"fireworks_ai",
"groq",

View file

View file

@ -0,0 +1,925 @@
"""Tests for the ClinePass provider.
The point of the end-to-end tests here is that they drive ``litellm.completion()``
with a mocked transport rather than calling the transforms directly -- a unit test
that calls ``transform_response()`` itself proves the function is correct but not
that anything invokes it, which is exactly how the envelope unwrap was previously
shipped as dead code.
"""
import json
from typing import Final
from unittest.mock import patch
import httpx
import pytest
from starlette.websockets import WebSocket
import litellm
from litellm.litellm_core_utils.litellm_logging import Logging
from litellm.llms.clinepass.chat.transformation import (
ClinePassConfig,
_apply_model_prefix,
_unwrap_response_envelope,
)
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler
from litellm.proxy._types import UserAPIKeyAuth
from litellm.responses.main import _aresponses_websocket
from litellm.types.utils import LlmProviders
from litellm.utils import ProviderConfigManager
API_KEY = "sk-clinepass-test-not-real"
ENVELOPED_COMPLETION = {
"success": True,
"data": {
"id": "chatcmpl-test",
"object": "chat.completion",
"created": 1,
"model": "clinepass/deepseek-v4-flash",
"choices": [
{
"index": 0,
"finish_reason": "stop",
"message": {
"role": "assistant",
"content": "pong",
"reasoning": "the user asked for pong",
},
}
],
"usage": {"prompt_tokens": 5, "completion_tokens": 2, "total_tokens": 7},
},
}
def _response(payload: dict, url: str = "https://api.cline.bot/api/v1/chat/completions") -> httpx.Response:
return httpx.Response(200, json=payload, request=httpx.Request("POST", url))
@pytest.fixture(autouse=True)
def _clinepass_env(monkeypatch):
monkeypatch.setenv("CLINEPASS_API_KEY", API_KEY)
monkeypatch.delenv("CLINEPASS_API_BASE", raising=False)
@pytest.fixture
def unrelated_credentials(monkeypatch):
for env_name in ("OPENAI_API_KEY", "ANTHROPIC_API_KEY", "GROQ_API_KEY", "OPENROUTER_API_KEY"):
monkeypatch.setenv(env_name, f"sk-unrelated-{env_name}")
for attribute in ("api_key", "openai_key", "anthropic_key"):
monkeypatch.setattr(litellm, attribute, f"sk-unrelated-{attribute}")
@pytest.mark.parametrize("credential_source", ["missing", "environment", "explicit"])
@pytest.mark.parametrize("custom_llm_provider", [None, "clinepass"])
def test_chat_sends_only_clinepass_credentials(
monkeypatch, unrelated_credentials, credential_source, custom_llm_provider
):
if credential_source == "missing":
monkeypatch.delenv("CLINEPASS_API_KEY", raising=False)
explicit_key: Final = "cp-request-key" if credential_source == "explicit" else None
captured = {}
def fake_post(self, url, *args, **kwargs):
captured["url"] = str(url)
captured["headers"] = httpx.Headers(kwargs["headers"])
return _response(ENVELOPED_COMPLETION)
with patch.object(HTTPHandler, "post", fake_post):
response = litellm.completion(
model="deepseek-v4-flash" if custom_llm_provider else "clinepass/deepseek-v4-flash",
custom_llm_provider=custom_llm_provider,
api_key=explicit_key,
messages=[{"role": "user", "content": "ping"}],
)
expected_key: Final = explicit_key or (API_KEY if credential_source == "environment" else None)
assert captured["url"] == "https://api.cline.bot/api/v1/chat/completions"
assert captured["headers"].get("authorization") == (f"Bearer {expected_key}" if expected_key else None)
assert response.choices[0].message.content == "pong"
@pytest.mark.parametrize("credential_source", ["missing", "environment", "explicit"])
@pytest.mark.asyncio
async def test_async_chat_sends_only_clinepass_credentials(monkeypatch, unrelated_credentials, credential_source):
if credential_source == "missing":
monkeypatch.delenv("CLINEPASS_API_KEY", raising=False)
explicit_key: Final = "cp-request-key" if credential_source == "explicit" else None
captured = {}
async def fake_post(self, url, *args, **kwargs):
captured["url"] = str(url)
captured["headers"] = httpx.Headers(kwargs["headers"])
return _response(ENVELOPED_COMPLETION)
with patch.object(AsyncHTTPHandler, "post", fake_post):
response = await litellm.acompletion(
model="clinepass/deepseek-v4-flash",
api_key=explicit_key,
messages=[{"role": "user", "content": "ping"}],
)
expected_key: Final = explicit_key or (API_KEY if credential_source == "environment" else None)
assert captured["url"] == "https://api.cline.bot/api/v1/chat/completions"
assert captured["headers"].get("authorization") == (f"Bearer {expected_key}" if expected_key else None)
assert response.choices[0].message.content == "pong"
@pytest.mark.parametrize("credential_source", ["missing", "environment", "explicit"])
@pytest.mark.parametrize("connection_provider", ["clinepass", "mistral"])
@pytest.mark.parametrize("changed_model", [None, "openai/gpt-4o", "clinepass/unauthorized-model"])
@pytest.mark.asyncio
async def test_managed_responses_websocket_sends_only_clinepass_credentials(
monkeypatch, unrelated_credentials, credential_source, connection_provider, changed_model
):
if credential_source == "missing":
monkeypatch.delenv("CLINEPASS_API_KEY", raising=False)
connection_key: Final = (
"sk-unrelated-mistral"
if connection_provider == "mistral"
else "cp-request-key"
if credential_source == "explicit"
else None
)
sent = []
received = []
foreign_requests = []
lifecycle = iter(
(
{"type": "websocket.connect"},
*(
(
{
"type": "websocket.receive",
"text": json.dumps({"type": "response.create", "model": changed_model, "input": "hi"}),
},
)
if changed_model is not None
else ()
),
{"type": "websocket.disconnect", "code": 1000},
)
)
async def block_foreign_request(self, request, *args, **kwargs):
foreign_requests.append(str(request.url))
raise AssertionError("Unexpected provider HTTP request")
monkeypatch.setattr(httpx.AsyncClient, "send", block_foreign_request)
async def receive():
return next(lifecycle)
async def send(message):
if message["type"] == "websocket.send":
received.append(json.loads(message["text"]))
websocket: Final = WebSocket(
scope={"type": "websocket", "path": "/v1/responses", "headers": [], "query_string": b""},
receive=receive,
send=send,
)
await websocket.accept()
chunk: Final = {
"id": "chatcmpl-test",
"object": "chat.completion.chunk",
"created": 1,
"model": "cline-pass/deepseek-v4-flash",
"choices": [{"index": 0, "delta": {"role": "assistant", "content": "pong"}, "finish_reason": "stop"}],
}
async def fake_post(self, url, *args, **kwargs):
sent.append((str(url), httpx.Headers(kwargs["headers"]).get("authorization")))
return httpx.Response(
200,
content=f"data: {json.dumps(chunk)}\n\ndata: [DONE]\n\n".encode(),
headers={"content-type": "text/event-stream"},
request=httpx.Request("POST", str(url)),
)
with patch.object(AsyncHTTPHandler, "post", fake_post):
result = await _aresponses_websocket.__wrapped__(
model=f"{connection_provider}/deepseek-v4-flash",
websocket=websocket,
api_key=connection_key,
user_api_key_dict=(
UserAPIKeyAuth(models=[f"{connection_provider}/deepseek-v4-flash"])
if changed_model is not None
else None
),
first_message=json.dumps(
{"type": "response.create", "model": "clinepass/deepseek-v4-flash", "input": "ping"}
),
litellm_logging_obj=Logging(
model=f"{connection_provider}/deepseek-v4-flash",
messages=[],
stream=True,
call_type="aresponses",
start_time=0,
litellm_call_id="cp-ws-test",
function_id="cp-ws-test",
),
)
expected_key: Final = (
connection_key
if connection_provider == "clinepass" and credential_source == "explicit"
else API_KEY
if credential_source != "missing"
else None
)
assert sent == (
[]
if changed_model is not None and connection_provider == "mistral"
else [("https://api.cline.bot/api/v1/chat/completions", f"Bearer {expected_key}" if expected_key else None)]
)
assert result is None
errors: Final = [event["error"] for event in received if event["type"] == "error"]
assert errors == (
[
{
"type": "invalid_request_error",
"message": "Changing models requires a new authorized WebSocket connection",
}
]
* (2 if connection_provider == "mistral" else 1)
if changed_model is not None
else []
)
assert foreign_requests == []
assert ("response.completed" in [event["type"] for event in received]) == bool(sent)
# --------------------------------------------------------------------------
# Registration / routing
# --------------------------------------------------------------------------
def test_get_llm_provider_resolves_clinepass():
model, provider, api_key, api_base = litellm.get_llm_provider(model="clinepass/deepseek-v4-flash")
assert model == "deepseek-v4-flash"
assert provider == "clinepass"
assert api_key == API_KEY
assert api_base == "https://api.cline.bot/api/v1"
def test_provider_config_manager_returns_clinepass_config():
config = ProviderConfigManager.get_provider_chat_config(model="deepseek-v4-flash", provider=LlmProviders.CLINEPASS)
assert isinstance(config, ClinePassConfig)
def test_clinepass_is_not_a_json_configured_provider_via_behaviour():
"""ClinePass needs a response transform, which the JSON provider system's
OpenAI-SDK dispatch path never invokes. We assert it is not on that path
by verifying the envelope unwrap actually triggers."""
captured = {}
def fake_post(self, url, *args, **kwargs):
captured["body"] = json.loads(kwargs["data"])
return _response(ENVELOPED_COMPLETION)
with patch.object(HTTPHandler, "post", fake_post):
response = litellm.completion(
model="clinepass/deepseek-v4-flash",
messages=[{"role": "user", "content": "ping"}],
)
# The unwrap works, meaning we didn't drift into the JSON provider registry
# which would have bypassed our custom transform.
assert response.choices[0].message.content == "pong"
def test_clinepass_is_not_in_openai_compatible_providers():
"""The cheap structural guard for the credential leak. Keep it.
The behavioural test below proves the *consequence*; this proves the
*cause*, in one line and with no mocking that could itself be wrong. Both
are wanted: a mocked behavioural test can drift into passing for the wrong
reason, while this cannot.
The membership is not routing-inert, which is what made it dangerous. The
list is also read by the speech branch in `main.py` -- which sends to the
provider's own `api_base` while taking the key from `OPENAI_API_KEY`, so a
`litellm.speech(model="clinepass/...")` call shipped the caller's OpenAI
credential to the Cline host for an endpoint ClinePass does not implement --
by `OPENAI_AUDIO_TRANSCRIPTION_PROVIDERS`, which is derived from it, by
image generation in `litellm/images/main.py`, and by
`_add_provider_specific_params`, which nests unknown kwargs under
`extra_body` for listed providers. `BaseLLMHTTPHandler` merges that back into
the request body, so the wire body is the same either way: membership is
pinned structurally here and at the params level in
`test_unknown_kwargs_are_flattened_not_wrapped_in_extra_body`, not by the
request body.
Exception mapping is preserved by registering ClinePass explicitly beside
`mistral` in `exception_mapping_utils.py`; see
`test_upstream_401_maps_to_authentication_error`.
"""
assert "clinepass" not in litellm.openai_compatible_providers
assert "clinepass" not in litellm.constants.OPENAI_AUDIO_TRANSCRIPTION_PROVIDERS
def test_clinepass_is_not_in_openai_compatible_providers_via_behaviour():
"""ClinePass must NOT be in `openai_compatible_providers`.
The drift guard here is the transcription half: a listed provider is picked up
by the OpenAI transcription branch instead of raising as unmapped.
The body assertions only pin the contract that an unknown kwarg reaches the
JSON body flat; they hold for listed providers too, so they do not detect
drift (the request body is identical either way)."""
captured = {}
def fake_post(self, url, *args, **kwargs):
captured["body"] = json.loads(kwargs["data"])
return _response(ENVELOPED_COMPLETION)
with patch.object(HTTPHandler, "post", fake_post):
litellm.completion(
model="clinepass/deepseek-v4-flash",
messages=[{"role": "user", "content": "ping"}],
custom_vendor_flag=True, # unknown kwarg
)
assert "extra_body" not in captured["body"]
assert captured["body"].get("custom_vendor_flag") is True
# Audio transcription should outright fail as unmapped, confirming it
# isn't implicitly picked up by OPENAI_AUDIO_TRANSCRIPTION_PROVIDERS
with pytest.raises(ValueError, match="Unmapped provider"):
litellm.transcription(
model="clinepass/deepseek-v4-flash",
file=b"fake audio data",
)
def test_api_base_env_override(monkeypatch):
monkeypatch.setenv("CLINEPASS_API_BASE", "https://proxy.internal/api/v1")
_, _, _, api_base = litellm.get_llm_provider(model="clinepass/deepseek-v4-flash")
assert api_base == "https://proxy.internal/api/v1"
@pytest.mark.parametrize(
"api_base,expected",
[
(None, "https://api.cline.bot/api/v1/chat/completions"),
("https://api.cline.bot/api/v1", "https://api.cline.bot/api/v1/chat/completions"),
("https://api.cline.bot/api/v1/", "https://api.cline.bot/api/v1/chat/completions"),
(
"https://api.cline.bot/api/v1/chat/completions",
"https://api.cline.bot/api/v1/chat/completions",
),
],
)
def test_get_complete_url(api_base, expected):
url = ClinePassConfig().get_complete_url(
api_base=api_base, api_key=API_KEY, model="deepseek-v4-flash", optional_params={}, litellm_params={}
)
assert url == expected
# --------------------------------------------------------------------------
# Model prefix
# --------------------------------------------------------------------------
def test_model_prefix_restored_on_bare_id():
"""The restored qualifier is the catalog namespace ``cline-pass/`` (hyphenated),
NOT LiteLLM's own ``clinepass/`` routing prefix.
The API validates only the *shape* of a model id, so a wrong namespace still
returns HTTP 200 -- but it does not always resolve to the same underlying
model, which makes a wrong value silent rather than harmless.
"""
assert _apply_model_prefix({"model": "deepseek-v4-flash"})["model"] == "cline-pass/deepseek-v4-flash"
def test_model_prefix_left_alone_when_qualifier_present():
"""`clinepass/openrouter/foo` arrives here as `openrouter/foo` and must pass through."""
assert _apply_model_prefix({"model": "openrouter/foo"})["model"] == "openrouter/foo"
def test_model_prefix_ignores_missing_model():
assert _apply_model_prefix({}) == {}
# --------------------------------------------------------------------------
# Response envelope
# --------------------------------------------------------------------------
def test_unwrap_envelope_extracts_inner_completion():
unwrapped = _unwrap_response_envelope(_response(ENVELOPED_COMPLETION))
assert unwrapped.json() == ENVELOPED_COMPLETION["data"]
def test_unwrap_envelope_content_length_describes_the_new_body():
"""The original content-length describes the enveloped bytes and must not be
carried over; httpx recomputes a correct one for the rewritten body."""
raw = _response(ENVELOPED_COMPLETION)
unwrapped = _unwrap_response_envelope(raw)
assert unwrapped.headers["content-length"] != raw.headers["content-length"]
assert int(unwrapped.headers["content-length"]) == len(unwrapped.content)
def test_unwrap_envelope_passes_through_openai_shaped_body():
payload = ENVELOPED_COMPLETION["data"]
assert _unwrap_response_envelope(_response(payload)).json() == payload
def test_unwrap_envelope_passes_through_error_nested_under_same_key():
"""An error under `data` has no `choices` and must not be mistaken for a completion."""
payload = {"success": False, "data": {"message": "bad model"}}
assert _unwrap_response_envelope(_response(payload)).json() == payload
def test_unwrap_envelope_passes_through_non_json_body():
raw = httpx.Response(
200, content=b"not json", request=httpx.Request("POST", "https://api.cline.bot/api/v1/chat/completions")
)
assert _unwrap_response_envelope(raw) is raw
# --------------------------------------------------------------------------
# Parameter mapping
# --------------------------------------------------------------------------
def test_max_completion_tokens_mapped_to_max_tokens():
mapped = ClinePassConfig().map_openai_params(
non_default_params={"max_completion_tokens": 4000},
optional_params={},
model="deepseek-v4-flash",
drop_params=False,
)
assert mapped == {"max_tokens": 4000}
# --------------------------------------------------------------------------
# End-to-end through litellm.completion() -- these are the load-bearing ones
# --------------------------------------------------------------------------
def test_completion_unwraps_envelope_and_prefixes_model():
captured = {}
def fake_post(self, url, *args, **kwargs):
captured["url"] = str(url)
captured["body"] = json.loads(kwargs["data"])
return _response(ENVELOPED_COMPLETION)
with patch.object(HTTPHandler, "post", fake_post):
response = litellm.completion(
model="clinepass/deepseek-v4-flash",
messages=[{"role": "user", "content": "ping"}],
max_tokens=4000,
)
assert captured["url"] == "https://api.cline.bot/api/v1/chat/completions"
assert captured["body"]["model"] == "cline-pass/deepseek-v4-flash"
assert response.choices[0].message.content == "pong"
assert response.choices[0].message.reasoning_content == "the user asked for pong"
@pytest.mark.asyncio
async def test_acompletion_unwraps_envelope_and_prefixes_model():
captured = {}
async def fake_post(self, url, *args, **kwargs):
captured["url"] = str(url)
captured["body"] = json.loads(kwargs["data"])
return _response(ENVELOPED_COMPLETION)
with patch.object(AsyncHTTPHandler, "post", fake_post):
response = await litellm.acompletion(
model="clinepass/deepseek-v4-flash",
messages=[{"role": "user", "content": "ping"}],
max_tokens=4000,
)
assert captured["url"] == "https://api.cline.bot/api/v1/chat/completions"
assert captured["body"]["model"] == "cline-pass/deepseek-v4-flash"
assert response.choices[0].message.content == "pong"
@pytest.mark.parametrize("credential_source", ["missing", "environment", "explicit"])
def test_completion_streaming_is_not_unwrapped(monkeypatch, unrelated_credentials, credential_source):
"""ClinePass does NOT wrap SSE chunks -- they are already OpenAI-shaped."""
if credential_source == "missing":
monkeypatch.delenv("CLINEPASS_API_KEY", raising=False)
explicit_key: Final = "cp-stream-key" if credential_source == "explicit" else None
captured = {}
chunks = [
{
"id": "chatcmpl-test",
"object": "chat.completion.chunk",
"created": 1,
"model": "clinepass/deepseek-v4-flash",
"choices": [{"index": 0, "delta": {"content": piece}, "finish_reason": None}],
}
for piece in ["one ", "two ", "three"]
]
chunks.append(
{
"id": "chatcmpl-test",
"object": "chat.completion.chunk",
"created": 1,
"model": "clinepass/deepseek-v4-flash",
"choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}],
}
)
body = "".join(f"data: {json.dumps(c)}\n\n" for c in chunks) + "data: [DONE]\n\n"
def fake_post(self, url, *args, **kwargs):
captured["authorization"] = httpx.Headers(kwargs["headers"]).get("authorization")
return httpx.Response(
200,
content=body.encode(),
headers={"content-type": "text/event-stream"},
request=httpx.Request("POST", str(url)),
)
with patch.object(HTTPHandler, "post", fake_post):
stream = litellm.completion(
model="clinepass/deepseek-v4-flash",
messages=[{"role": "user", "content": "count"}],
api_key=explicit_key,
max_tokens=4000,
stream=True,
)
text = ""
finish_reason = None
for c in stream:
if c.choices:
if c.choices[0].delta.content:
text += c.choices[0].delta.content
if c.choices[0].finish_reason:
finish_reason = c.choices[0].finish_reason
assert text == "one two three"
assert finish_reason == "stop"
expected_key: Final = explicit_key or (API_KEY if credential_source == "environment" else None)
assert captured["authorization"] == (f"Bearer {expected_key}" if expected_key else None)
@pytest.mark.parametrize("credential_source", ["missing", "environment", "explicit"])
@pytest.mark.asyncio
async def test_acompletion_streaming_is_not_unwrapped(monkeypatch, unrelated_credentials, credential_source):
if credential_source == "missing":
monkeypatch.delenv("CLINEPASS_API_KEY", raising=False)
explicit_key: Final = "cp-stream-key" if credential_source == "explicit" else None
captured = {}
chunks = [
{
"id": "chatcmpl-test",
"object": "chat.completion.chunk",
"created": 1,
"model": "clinepass/deepseek-v4-flash",
"choices": [{"index": 0, "delta": {"content": piece}, "finish_reason": None}],
}
for piece in ["one ", "two ", "three"]
]
chunks.append(
{
"id": "chatcmpl-test",
"object": "chat.completion.chunk",
"created": 1,
"model": "clinepass/deepseek-v4-flash",
"choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}],
}
)
body = "".join(f"data: {json.dumps(c)}\n\n" for c in chunks) + "data: [DONE]\n\n"
async def fake_post(self, url, *args, **kwargs):
captured["authorization"] = httpx.Headers(kwargs["headers"]).get("authorization")
return httpx.Response(
200,
content=body.encode(),
headers={"content-type": "text/event-stream"},
request=httpx.Request("POST", str(url)),
)
with patch.object(AsyncHTTPHandler, "post", fake_post):
stream = await litellm.acompletion(
model="clinepass/deepseek-v4-flash",
messages=[{"role": "user", "content": "count"}],
api_key=explicit_key,
max_tokens=4000,
stream=True,
)
text = ""
finish_reason = None
async for c in stream:
if c.choices:
if c.choices[0].delta.content:
text += c.choices[0].delta.content
if c.choices[0].finish_reason:
finish_reason = c.choices[0].finish_reason
assert text == "one two three"
assert finish_reason == "stop"
expected_key: Final = explicit_key or (API_KEY if credential_source == "environment" else None)
assert captured["authorization"] == (f"Bearer {expected_key}" if expected_key else None)
def test_completion_streaming_tool_call_reassembly():
"""Tool calls split across chunks must be correctly passed through by the OpenAI-compatible stream processor."""
chunks = [
{
"id": "chatcmpl-test",
"object": "chat.completion.chunk",
"created": 1,
"model": "clinepass/deepseek-v4-flash",
"choices": [
{
"index": 0,
"delta": {
"tool_calls": [
{
"index": 0,
"id": "call_123",
"type": "function",
"function": {"name": "get_weather", "arguments": ""},
}
]
},
"finish_reason": None,
}
],
},
{
"id": "chatcmpl-test",
"object": "chat.completion.chunk",
"created": 1,
"model": "clinepass/deepseek-v4-flash",
"choices": [
{
"index": 0,
"delta": {"tool_calls": [{"index": 0, "function": {"arguments": '{"loc'}}]},
"finish_reason": None,
}
],
},
{
"id": "chatcmpl-test",
"object": "chat.completion.chunk",
"created": 1,
"model": "clinepass/deepseek-v4-flash",
"choices": [
{
"index": 0,
"delta": {"tool_calls": [{"index": 0, "function": {"arguments": 'ation": "NYC"}'}}]},
"finish_reason": None,
}
],
},
{
"id": "chatcmpl-test",
"object": "chat.completion.chunk",
"created": 1,
"model": "clinepass/deepseek-v4-flash",
"choices": [{"index": 0, "delta": {}, "finish_reason": "tool_calls"}],
},
]
body = "".join(f"data: {json.dumps(c)}\n\n" for c in chunks) + "data: [DONE]\n\n"
def fake_post(self, url, *args, **kwargs):
return httpx.Response(
200,
content=body.encode(),
headers={"content-type": "text/event-stream"},
request=httpx.Request("POST", str(url)),
)
with patch.object(HTTPHandler, "post", fake_post):
stream = litellm.completion(
model="clinepass/deepseek-v4-flash",
messages=[{"role": "user", "content": "weather"}],
stream=True,
)
args_text = ""
for c in stream:
if c.choices and c.choices[0].delta.tool_calls:
tc = c.choices[0].delta.tool_calls[0]
if tc.function and tc.function.arguments:
args_text += tc.function.arguments
assert args_text == '{"location": "NYC"}'
def test_completion_preserves_usage():
def fake_post(self, url, *args, **kwargs):
return _response(ENVELOPED_COMPLETION)
with patch.object(HTTPHandler, "post", fake_post):
response = litellm.completion(
model="clinepass/deepseek-v4-flash",
messages=[{"role": "user", "content": "ping"}],
)
assert response.usage.prompt_tokens == 5
assert response.usage.completion_tokens == 2
assert response.usage.total_tokens == 7
def test_completion_sends_authorization_header():
captured_headers = {}
def fake_post(self, url, *args, **kwargs):
captured_headers.update(kwargs.get("headers", {}))
return _response(ENVELOPED_COMPLETION)
with patch.object(HTTPHandler, "post", fake_post):
litellm.completion(
model="clinepass/deepseek-v4-flash",
messages=[{"role": "user", "content": "ping"}],
)
assert captured_headers.get("Authorization") == f"Bearer {API_KEY}"
def test_completion_explicit_api_key_precedence():
captured_headers = {}
def fake_post(self, url, *args, **kwargs):
captured_headers.update(kwargs.get("headers", {}))
return _response(ENVELOPED_COMPLETION)
with patch.object(HTTPHandler, "post", fake_post):
litellm.completion(
model="clinepass/deepseek-v4-flash",
messages=[{"role": "user", "content": "ping"}],
api_key="sk-clinepass-explicit-key",
)
assert captured_headers.get("Authorization") == "Bearer sk-clinepass-explicit-key"
def test_upstream_429_maps_to_rate_limit_error():
from litellm.exceptions import RateLimitError
def fake_post(self, url, *args, **kwargs):
raise httpx.HTTPStatusError(
"Too Many Requests",
request=httpx.Request("POST", str(url)),
response=httpx.Response(
429,
json={"error": "Rate limit exceeded"},
request=httpx.Request("POST", str(url)),
),
)
with patch.object(HTTPHandler, "post", fake_post), pytest.raises(RateLimitError) as excinfo:
litellm.completion(
model="clinepass/deepseek-v4-flash",
messages=[{"role": "user", "content": "hi"}],
)
assert excinfo.value.status_code == 429
def test_upstream_401_maps_to_authentication_error():
"""ClinePass answers a bad key with HTTP 401; that must not be flattened
into a generic APIConnectionError."""
from litellm.exceptions import AuthenticationError
def fake_post(self, url, *args, **kwargs):
raise httpx.HTTPStatusError(
"Unauthorized",
request=httpx.Request("POST", str(url)),
response=httpx.Response(
401,
json={"error": "Unauthorized"},
request=httpx.Request("POST", str(url)),
),
)
with patch.object(HTTPHandler, "post", fake_post), pytest.raises(AuthenticationError) as excinfo:
litellm.completion(
model="clinepass/deepseek-v4-flash",
messages=[{"role": "user", "content": "hi"}],
max_tokens=100,
)
assert excinfo.value.status_code == 401
# --------------------------------------------------------------------------
# Truncation reporting
#
# ClinePass was once observed returning finish_reason "stop" on a completion cut
# off by max_tokens. Re-probing the live API on 2026-08-22 could not reproduce
# it, so the provider now reports the upstream finish reason unmodified to avoid
# false positives on natural completions that land exactly on the cap.
# --------------------------------------------------------------------------
def _truncated_envelope(completion_tokens: int, finish_reason: str = "stop") -> dict:
payload = json.loads(json.dumps(ENVELOPED_COMPLETION))
payload["data"]["choices"][0]["finish_reason"] = finish_reason
payload["data"]["usage"]["completion_tokens"] = completion_tokens
return payload
def _complete(payload: dict, **kwargs):
def fake_post(self, url, *args, **post_kwargs):
return _response(payload)
with patch.object(HTTPHandler, "post", fake_post):
return litellm.completion(
model="clinepass/deepseek-v4-flash",
messages=[{"role": "user", "content": "ping"}],
**kwargs,
)
def test_completion_at_the_cap_preserves_upstream_stop():
response = _complete(_truncated_envelope(4000), max_tokens=4000)
assert response.choices[0].finish_reason == "stop"
def test_completion_over_the_cap_preserves_upstream_stop():
response = _complete(_truncated_envelope(4001), max_tokens=4000)
assert response.choices[0].finish_reason == "stop"
def test_completion_below_the_cap_keeps_stop():
response = _complete(_truncated_envelope(3999), max_tokens=4000)
assert response.choices[0].finish_reason == "stop"
def test_upstream_length_is_left_alone():
response = _complete(_truncated_envelope(4000, finish_reason="length"), max_tokens=4000)
assert response.choices[0].finish_reason == "length"
def test_no_max_tokens_means_no_rewrite():
response = _complete(_truncated_envelope(4000))
assert response.choices[0].finish_reason == "stop"
def test_max_completion_tokens_param_preserves_upstream_stop():
"""``max_completion_tokens`` is mapped to ``max_tokens`` before ``request_data`` is built."""
response = _complete(_truncated_envelope(4000), max_completion_tokens=4000)
assert response.choices[0].finish_reason == "stop"
# --------------------------------------------------------------------------
# Model catalog
# --------------------------------------------------------------------------
def test_get_models_returns_empty_without_calling_the_api():
"""ClinePass has no /models endpoint (404), and the inherited OpenAI
implementation would ask for it at the wrong path. It must not make the
request at all."""
def explode(*args, **kwargs): # pragma: no cover - must never run
raise AssertionError("get_models() must not perform an HTTP request")
with patch.object(litellm.module_level_client, "get", explode):
assert ClinePassConfig().get_models(api_key=API_KEY) == []
# --------------------------------------------------------------------------
# httpx internals
# --------------------------------------------------------------------------
def test_unwrap_envelope_survives_a_response_with_no_request_attached():
"""`httpx.Response.request` RAISES RuntimeError rather than returning None
when no request is attached, so the unwrap must ask for it defensively."""
raw = httpx.Response(200, json=ENVELOPED_COMPLETION)
unwrapped = _unwrap_response_envelope(raw)
assert unwrapped.json() == ENVELOPED_COMPLETION["data"]
@pytest.mark.asyncio
async def test_acompletion_uses_sync_transform_request_via_behaviour():
"""BaseLLMHTTPHandler builds the body with the sync transform_request on both
paths, so an async override would be dead code -- the shape of bug this
provider already shipped once. We assert this by verifying `acompletion`
invokes the sync transform (which we mock here to prove it runs)."""
captured = {}
# We patch the sync transform_request to prove it is the one called
# during the async flow.
original_transform = ClinePassConfig().transform_request
def mock_transform_request(*args, **kwargs):
captured["called"] = True
return original_transform(*args, **kwargs)
async def fake_post(self, url, *args, **kwargs):
return _response(ENVELOPED_COMPLETION)
with (
patch.object(ClinePassConfig, "transform_request", side_effect=mock_transform_request),
patch.object(AsyncHTTPHandler, "post", fake_post),
):
await litellm.acompletion(
model="clinepass/deepseek-v4-flash",
messages=[{"role": "user", "content": "ping"}],
)
assert captured.get("called") is True

View file

@ -0,0 +1,259 @@
"""ClinePass must not be reachable on endpoints it does not implement.
ClinePass implements chat completions only. Before this guard existed, listing
it in `litellm.openai_compatible_providers` made the speech, transcription and
image-generation branches in litellm match it. Those branches send the request
to the provider's own `api_base` but read the credential from `OPENAI_API_KEY`,
so a `litellm.speech(model="clinepass/...")` call POSTed the caller's OpenAI key
to the Cline host -- for an endpoint that does not exist there.
Asserting "not in the list" is a structural check and lives with the other
registry tests. These tests assert the behaviour instead: that no HTTP request
leaves the process at all, and that the OpenAI credential is never transmitted.
"""
import contextlib
from typing import Final
import httpx
import pytest
from openai import AsyncOpenAI, OpenAI
import litellm
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler
SENTINEL_OPENAI_KEY = "sk-sentinel-openai-key-must-never-be-transmitted"
@pytest.fixture
def no_request_allowed(monkeypatch):
"""Fail loudly if anything attempts an outbound request, and record it.
Patched at the httpx transport layer rather than at litellm's handlers, so a
future dispatch path that bypasses HTTPHandler is still caught.
"""
attempted: list[tuple[str, str]] = []
def record_and_block(self, request, *args, **kwargs):
attempted.append((str(request.url), request.headers.get("authorization", "")))
raise AssertionError(f"outbound request attempted to {request.url}")
def record_handler_and_block(self, url, *args, **kwargs):
attempted.append((str(url), httpx.Headers(kwargs.get("headers") or {}).get("authorization", "")))
raise AssertionError(f"outbound request attempted to {url}")
async def record_async_handler_and_block(self, url, *args, **kwargs):
record_handler_and_block(self, url, *args, **kwargs)
monkeypatch.setattr(httpx.Client, "send", record_and_block, raising=True)
monkeypatch.setattr(httpx.AsyncClient, "send", record_and_block, raising=True)
monkeypatch.setattr(HTTPHandler, "post", record_handler_and_block, raising=True)
monkeypatch.setattr(AsyncHTTPHandler, "post", record_async_handler_and_block, raising=True)
monkeypatch.setenv("OPENAI_API_KEY", SENTINEL_OPENAI_KEY)
monkeypatch.setenv("CLINEPASS_API_KEY", "cp-test-key")
return attempted
def test_speech_makes_no_outbound_request(no_request_allowed):
# Which exception litellm raises for an unsupported endpoint is its business
# and may change; that nothing is transmitted is the contract under test. The
# fixture's AssertionError means the network WAS reached, so it must escape
# rather than be swallowed as "some exception happened".
try:
litellm.speech(model="clinepass/deepseek-v4-flash", input="hi", voice="alloy")
except AssertionError:
raise
except Exception: # noqa: S110 - deliberate; see above
pass
assert no_request_allowed == []
@pytest.mark.parametrize("clinepass_key", [None, "cp-test-key"])
@pytest.mark.parametrize("custom_llm_provider", [None, "clinepass"])
@pytest.mark.parametrize("explicit_key", [None, "cp-explicit-key"])
def test_moderation_rejects_clinepass_before_credential_fallback(
monkeypatch, no_request_allowed, clinepass_key, custom_llm_provider, explicit_key
):
if clinepass_key is None:
monkeypatch.delenv("CLINEPASS_API_KEY", raising=False)
monkeypatch.setattr(litellm, "api_key", SENTINEL_OPENAI_KEY)
model: Final = "deepseek-v4-flash" if custom_llm_provider else "clinepass/deepseek-v4-flash"
with pytest.raises(litellm.BadRequestError, match="ClinePass does not support moderation endpoints"):
litellm.moderation(model=model, input="hi", custom_llm_provider=custom_llm_provider, api_key=explicit_key)
assert no_request_allowed == []
@pytest.mark.parametrize("clinepass_key", [None, "cp-test-key"])
@pytest.mark.parametrize("custom_llm_provider", [None, "clinepass"])
@pytest.mark.parametrize("explicit_key", [None, "cp-explicit-key"])
@pytest.mark.asyncio
async def test_async_moderation_rejects_clinepass_before_credential_fallback(
monkeypatch, no_request_allowed, clinepass_key, custom_llm_provider, explicit_key
):
if clinepass_key is None:
monkeypatch.delenv("CLINEPASS_API_KEY", raising=False)
monkeypatch.setattr(litellm, "api_key", SENTINEL_OPENAI_KEY)
model: Final = "deepseek-v4-flash" if custom_llm_provider else "clinepass/deepseek-v4-flash"
with pytest.raises(litellm.BadRequestError, match="ClinePass does not support moderation endpoints"):
await litellm.amoderation(
model=model, input="hi", custom_llm_provider=custom_llm_provider, api_key=explicit_key
)
assert no_request_allowed == []
@pytest.mark.parametrize("async_mode", [False, True])
@pytest.mark.asyncio
async def test_openai_moderation_remains_supported(async_mode):
sent = []
def respond(request):
sent.append((str(request.url), request.headers["authorization"]))
return httpx.Response(
200,
json={
"id": "modr-test",
"model": "moderation-test",
"results": [{"flagged": False, "categories": {"violence": False}, "category_scores": {"violence": 0}}],
},
)
transport: Final = httpx.MockTransport(respond)
if async_mode:
async with AsyncOpenAI(
api_key="sk-moderation-test", http_client=httpx.AsyncClient(transport=transport)
) as client:
response = await litellm.amoderation(model="openai/moderation-test", input="hi", client=client)
else:
with OpenAI(api_key="sk-moderation-test", http_client=httpx.Client(transport=transport)) as client:
response = litellm.moderation(model="moderation-test", input="hi", client=client)
assert sent == [("https://api.openai.com/v1/moderations", "Bearer sk-moderation-test")]
assert response.id == "modr-test"
assert response.results[0].flagged is False
def test_transcription_makes_no_outbound_request(no_request_allowed, tmp_path):
audio = tmp_path / "a.mp3"
audio.write_bytes(b"\x00\x00")
with open(audio, "rb") as handle:
try:
litellm.transcription(model="clinepass/deepseek-v4-flash", file=handle)
except AssertionError:
raise
except Exception: # noqa: S110 - deliberate; see test_speech_makes_no_outbound_request
pass
assert no_request_allowed == []
def test_image_generation_makes_no_outbound_request(no_request_allowed):
"""Image generation must not reach the Cline host.
Note it does not raise either: litellm returns an empty `ImageResponse` for
any provider with no image support. That is pre-existing upstream behaviour,
not a ClinePass quirk -- `mistral`, which has the same shape ClinePass now
has (own module, absent from `openai_compatible_providers`, explicitly
registered for exception mapping), returns the same empty response with zero
outbound requests. So this test asserts the property that is ours to keep:
nothing is transmitted.
"""
litellm.image_generation(model="clinepass/deepseek-v4-flash", prompt="a cat")
assert no_request_allowed == []
def test_openai_credential_is_never_transmitted(no_request_allowed):
"""The point of the P1: whatever happens, the OpenAI key must not go out."""
for call in (
lambda: litellm.speech(model="clinepass/deepseek-v4-flash", input="hi", voice="alloy"),
lambda: litellm.transcription(model="clinepass/deepseek-v4-flash", file=None),
lambda: litellm.image_generation(model="clinepass/deepseek-v4-flash", prompt="a cat"),
):
# Whether each endpoint raises or returns an empty response is upstream's
# business; that no credential leaves the process is ours.
with contextlib.suppress(Exception):
call()
leaked = [url for url, auth in no_request_allowed if SENTINEL_OPENAI_KEY in auth]
assert leaked == [], f"OPENAI_API_KEY was transmitted to {leaked}"
def test_unknown_kwargs_are_flattened_not_wrapped_in_extra_body():
"""Unknown kwargs stay flat in the optional params.
For a provider listed in `openai_compatible_providers`, `get_optional_params`
nests them under `extra_body`. `BaseLLMHTTPHandler` later merges that back
into the request body, so the wire body does not distinguish the two cases;
this params-level check is what fails if ClinePass drifts back into the list.
"""
params = litellm.utils.get_optional_params(
model="cline-pass/deepseek-v4-flash",
custom_llm_provider="clinepass",
temperature=0.5,
some_vendor_knob=7,
)
assert "extra_body" not in params
assert params["some_vendor_knob"] == 7
def test_chat_does_not_fall_back_to_the_global_litellm_api_key(monkeypatch):
"""`litellm.api_key` is the caller's general-purpose (usually OpenAI) key.
With no ClinePass credential configured, chat must not borrow it: doing so
sends that key to the Cline host as a Bearer token.
"""
sent: list[tuple[str, str]] = []
def record_and_stop(self, *args, **kwargs):
sent.append((str(kwargs.get("url")), str((kwargs.get("headers") or {}).get("Authorization", ""))))
raise RuntimeError("stop before any network I/O")
monkeypatch.setattr(HTTPHandler, "post", record_and_stop, raising=True)
monkeypatch.delenv("CLINEPASS_API_KEY", raising=False)
monkeypatch.setattr(litellm, "api_key", SENTINEL_OPENAI_KEY)
with contextlib.suppress(Exception):
litellm.completion(
model="clinepass/deepseek-v4-flash",
messages=[{"role": "user", "content": "hi"}],
num_retries=0,
)
assert sent, "the chat request never reached the transport, so nothing was checked"
assert all(SENTINEL_OPENAI_KEY not in authorization for _, authorization in sent)
@pytest.mark.parametrize("clinepass_key", [None, "cp-test-key"])
@pytest.mark.parametrize("explicit_key", [None, "cp-explicit-key"])
@pytest.mark.parametrize("endpoint", ["client_secret", "transcription_session", "calls"])
@pytest.mark.asyncio
async def test_realtime_rejects_clinepass_before_credential_fallback(
monkeypatch, no_request_allowed, clinepass_key, explicit_key, endpoint
):
if clinepass_key is None:
monkeypatch.delenv("CLINEPASS_API_KEY", raising=False)
else:
monkeypatch.setenv("CLINEPASS_API_KEY", clinepass_key)
monkeypatch.setattr(litellm, "api_key", SENTINEL_OPENAI_KEY)
monkeypatch.setattr(litellm, "openai_key", SENTINEL_OPENAI_KEY)
kwargs: Final = {"model": "clinepass/deepseek-v4-flash", "api_key": explicit_key}
request: Final = (
litellm.acreate_realtime_client_secret(**kwargs)
if endpoint == "client_secret"
else litellm.acreate_realtime_transcription_session(**kwargs)
if endpoint == "transcription_session"
else litellm.arealtime_calls(openai_ephemeral_key=SENTINEL_OPENAI_KEY, sdp_body=b"v=0\r\n", **kwargs)
)
with pytest.raises(litellm.BadRequestError, match="ClinePass does not support realtime endpoints"):
await request
assert no_request_allowed == []

View file

@ -176,6 +176,7 @@ describe("provider_info_helpers", () => {
Providers.AUTO_ROUTER,
Providers.BYTEZ,
Providers.CLARIFAI,
Providers.CLINEPASS,
Providers.Cognition,
Providers.COMPACTIFAI,
Providers.DATAROBOT,

View file

@ -89,6 +89,7 @@ export enum Providers {
Cerebras = "Cerebras",
CHATGPT = "ChatGPT Subscription",
CLARIFAI = "Clarifai",
CLINEPASS = "ClinePass",
CLOUDFLARE = "Cloudflare",
CODESTRAL = "Codestral",
Cognition = "Cognition",
@ -208,6 +209,7 @@ export const provider_map: Record<string, string> = {
Cerebras: "cerebras",
CHATGPT: "chatgpt",
CLARIFAI: "clarifai",
CLINEPASS: "clinepass",
CLOUDFLARE: "cloudflare",
CODESTRAL: "codestral",
Cognition: "cognition",