mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-11 03:38:38 +00:00
Merge 0bc31c076f into 1d52985d03
This commit is contained in:
commit
cfa54fdb7e
25 changed files with 1626 additions and 10 deletions
|
|
@ -317,6 +317,7 @@ Set `LITELLM_PROXY_API_BASE` and `LITELLM_PROXY_API_KEY` and every model call th
|
|||
| [Bytez (`bytez`)](https://docs.litellm.ai/docs/providers/bytez) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Cerebras (`cerebras`)](https://docs.litellm.ai/docs/providers/cerebras) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Clarifai (`clarifai`)](https://docs.litellm.ai/docs/providers/clarifai) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [ClinePass (`clinepass`)](https://docs.litellm.ai/docs/providers/clinepass) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Cloudflare AI Workers (`cloudflare`)](https://docs.litellm.ai/docs/providers/cloudflare_workers) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Codestral (`codestral`)](https://docs.litellm.ai/docs/providers/codestral) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Cognition (`cognition`)](https://docs.litellm.ai/docs/providers/cognition) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
|
|
|
|||
|
|
@ -644,6 +644,7 @@ gemini_models: Set = set()
|
|||
xai_models: Set = set()
|
||||
zai_models: Set = set()
|
||||
deepseek_models: Set = set()
|
||||
clinepass_models: Set = set()
|
||||
tencent_models: Set = set()
|
||||
runwayml_models: Set = set()
|
||||
azure_ai_models: Set = set()
|
||||
|
|
@ -866,6 +867,8 @@ def _populate_provider_model_sets(model_cost_map: Dict) -> None:
|
|||
fal_ai_models.add(key)
|
||||
elif value.get("litellm_provider") == "deepseek":
|
||||
deepseek_models.add(key)
|
||||
elif value.get("litellm_provider") == "clinepass":
|
||||
clinepass_models.add(key) # pyright: ignore[reportUnknownArgumentType] # key comes from the untyped model cost map
|
||||
elif value.get("litellm_provider") == "tencent":
|
||||
tencent_models.add(key)
|
||||
elif value.get("litellm_provider") == "runwayml":
|
||||
|
|
@ -1078,6 +1081,7 @@ model_list = list(
|
|||
| zai_models
|
||||
| fal_ai_models
|
||||
| deepseek_models
|
||||
| clinepass_models
|
||||
| modelscope_models
|
||||
| azure_ai_models
|
||||
| voyage_models
|
||||
|
|
@ -1183,6 +1187,7 @@ def _build_models_by_provider() -> dict:
|
|||
"zai": zai_models,
|
||||
"fal_ai": fal_ai_models,
|
||||
"deepseek": deepseek_models,
|
||||
"clinepass": clinepass_models,
|
||||
"tencent": tencent_models,
|
||||
"runwayml": runwayml_models,
|
||||
"mistral": mistral_chat_models,
|
||||
|
|
@ -2052,6 +2057,9 @@ if TYPE_CHECKING:
|
|||
)
|
||||
from .llms.heroku.chat.transformation import HerokuChatConfig as HerokuChatConfig
|
||||
from .llms.cometapi.chat.transformation import CometAPIConfig as CometAPIConfig
|
||||
from .llms.clinepass.chat.transformation import (
|
||||
ClinePassConfig as ClinePassConfig,
|
||||
)
|
||||
from .llms.azure.chat.gpt_transformation import (
|
||||
AzureOpenAIConfig as AzureOpenAIConfig,
|
||||
)
|
||||
|
|
|
|||
|
|
@ -284,6 +284,7 @@ LLM_CONFIG_NAMES: Final = (
|
|||
"AzureOpenAIAssistantsAPIConfig",
|
||||
"HerokuChatConfig",
|
||||
"CometAPIConfig",
|
||||
"ClinePassConfig",
|
||||
"AzureOpenAIConfig",
|
||||
"AzureOpenAIGPT5Config",
|
||||
"AzureOpenAITextConfig",
|
||||
|
|
@ -1117,6 +1118,7 @@ _LLM_CONFIGS_IMPORT_MAP: Final = {
|
|||
),
|
||||
"HerokuChatConfig": (".llms.heroku.chat.transformation", "HerokuChatConfig"),
|
||||
"CometAPIConfig": (".llms.cometapi.chat.transformation", "CometAPIConfig"),
|
||||
"ClinePassConfig": (".llms.clinepass.chat.transformation", "ClinePassConfig"),
|
||||
"AzureOpenAIConfig": (".llms.azure.chat.gpt_transformation", "AzureOpenAIConfig"),
|
||||
"AzureOpenAIGPT5Config": (
|
||||
".llms.azure.chat.gpt_5_transformation",
|
||||
|
|
|
|||
|
|
@ -783,6 +783,7 @@ LITELLM_CHAT_PROVIDERS: Final = [
|
|||
"lemonade",
|
||||
"docker_model_runner",
|
||||
"amazon_nova",
|
||||
"clinepass",
|
||||
]
|
||||
|
||||
# Resolving these providers runs an OAuth device flow (their provider info IS the login), so any
|
||||
|
|
|
|||
|
|
@ -2468,6 +2468,7 @@ def exception_type(
|
|||
or custom_llm_provider == "text-completion-openai"
|
||||
or custom_llm_provider == "custom_openai"
|
||||
or custom_llm_provider in litellm.openai_compatible_providers
|
||||
or custom_llm_provider == "clinepass"
|
||||
or custom_llm_provider == "mistral"
|
||||
or custom_llm_provider == "runwayml"
|
||||
):
|
||||
|
|
|
|||
|
|
@ -611,6 +611,11 @@ def _get_openai_compatible_provider_info(
|
|||
api_base,
|
||||
dynamic_api_key,
|
||||
) = litellm.PerplexityChatConfig()._get_openai_compatible_provider_info(api_base, api_key)
|
||||
elif custom_llm_provider == "clinepass":
|
||||
(
|
||||
api_base,
|
||||
dynamic_api_key,
|
||||
) = litellm.ClinePassConfig()._get_openai_compatible_provider_info(api_base, api_key)
|
||||
elif custom_llm_provider == "aiohttp_openai":
|
||||
return model, "aiohttp_openai", api_key, api_base
|
||||
elif custom_llm_provider == "anyscale":
|
||||
|
|
|
|||
268
litellm/llms/clinepass/chat/transformation.py
Normal file
268
litellm/llms/clinepass/chat/transformation.py
Normal file
|
|
@ -0,0 +1,268 @@
|
|||
"""
|
||||
Support for ClinePass (the Cline API) `/v1/chat/completions` endpoint.
|
||||
|
||||
ClinePass is OpenAI-compatible apart from two quirks, both handled here:
|
||||
|
||||
1. Non-streaming completions are nested under a ``data`` envelope --
|
||||
``{"data": {"choices": [...]}, "success": true}`` -- rather than returning
|
||||
``choices`` at the top level. Streaming responses are *not* wrapped, so the
|
||||
inherited SSE handling needs no change.
|
||||
2. A bare model id is rejected with HTTP 400 ``invalid model format. Expected
|
||||
format: modelType/model``, but LiteLLM strips its own ``clinepass/`` routing
|
||||
prefix before the request is built, so a qualifier has to be restored.
|
||||
|
||||
Documentation: https://docs.cline.bot/
|
||||
|
||||
Credentials come only from the request's api_key or CLINEPASS_API_KEY.
|
||||
Moderation and realtime endpoints are unsupported and rejected before dispatch.
|
||||
"""
|
||||
|
||||
import json
|
||||
from typing import TYPE_CHECKING, Any, Final, NoReturn
|
||||
|
||||
import httpx
|
||||
|
||||
from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
||||
from litellm.secret_managers.main import get_secret_str
|
||||
from litellm.types.llms.openai import AllMessageValues
|
||||
from litellm.types.utils import ModelResponse
|
||||
|
||||
from ...openai.chat.gpt_transformation import OpenAIGPTConfig
|
||||
from ..common_utils import ClinePassException
|
||||
|
||||
# Mirrors litellm/llms/openai/chat/gpt_transformation.py: these are needed only
|
||||
# for annotations, and importing litellm_logging at runtime from a provider
|
||||
# module risks a circular import.
|
||||
if TYPE_CHECKING:
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as _LiteLLMLoggingObj
|
||||
from litellm.litellm_core_utils.tokenizer import Encoding as Tokenizer
|
||||
|
||||
LiteLLMLoggingObj = _LiteLLMLoggingObj
|
||||
else:
|
||||
LiteLLMLoggingObj = Any
|
||||
|
||||
CLINEPASS_API_BASE: Final = "https://api.cline.bot/api/v1"
|
||||
|
||||
# ClinePass nests the completion under this key on non-streaming responses.
|
||||
CLINEPASS_RESPONSE_ENVELOPE_KEY: Final = "data"
|
||||
|
||||
# The qualifier ClinePass expects on outbound model ids.
|
||||
#
|
||||
# Note the hyphen: the catalog namespace is ``cline-pass/``, not ``clinepass/``
|
||||
# (the latter is LiteLLM's own routing prefix, which is stripped before the
|
||||
# request is built). The API only validates the *shape* of a model id -- any
|
||||
# ``<segment>/<model>`` is accepted with HTTP 200 -- so an incorrect namespace
|
||||
# fails silently rather than loudly. It is not inert, though: for at least one
|
||||
# model an unrecognised namespace resolves to a different, date-pinned snapshot
|
||||
# (``cline-pass/deepseek-v4-flash`` -> ``deepseek/deepseek-v4-flash``, while
|
||||
# ``clinepass/deepseek-v4-flash`` -> ``deepseek/deepseek-v4-flash-0731``).
|
||||
CLINEPASS_MODEL_PREFIX: Final = "cline-pass/"
|
||||
|
||||
# Headers that describe the original byte stream and would be wrong once the
|
||||
# body is rewritten by _unwrap_response_envelope().
|
||||
_BODY_SPECIFIC_HEADERS: Final = ("content-length", "content-encoding")
|
||||
|
||||
|
||||
def _unwrap_response_envelope(raw_response: httpx.Response) -> httpx.Response:
|
||||
"""Strip ClinePass's ``data`` wrapper off a JSON completion body.
|
||||
|
||||
The OpenAI transforms read ``raw_response.json()`` directly, so rebuild the
|
||||
response around the inner object rather than duplicating their bodies here.
|
||||
|
||||
Returns the original response untouched whenever the body does not look like
|
||||
a wrapped completion, so an already-OpenAI-shaped body -- or an error nested
|
||||
under the same key -- is not mistaken for one.
|
||||
"""
|
||||
try:
|
||||
payload = raw_response.json()
|
||||
except ValueError:
|
||||
# Not a JSON body -- there is no envelope to strip.
|
||||
return raw_response
|
||||
|
||||
if not isinstance(payload, dict) or "choices" in payload:
|
||||
return raw_response
|
||||
|
||||
inner = payload.get(CLINEPASS_RESPONSE_ENVELOPE_KEY)
|
||||
if not isinstance(inner, dict) or "choices" not in inner:
|
||||
return raw_response
|
||||
|
||||
headers = {k: v for k, v in raw_response.headers.items() if k.lower() not in _BODY_SPECIFIC_HEADERS}
|
||||
|
||||
# httpx.Response.request raises RuntimeError rather than returning None when
|
||||
# no request is attached, so ask for it defensively instead of reaching for
|
||||
# the private attribute behind it.
|
||||
try:
|
||||
original_request = raw_response.request
|
||||
except RuntimeError:
|
||||
original_request = None
|
||||
|
||||
return httpx.Response(
|
||||
status_code=raw_response.status_code,
|
||||
headers=headers,
|
||||
content=json.dumps(inner, ensure_ascii=False).encode("utf-8"),
|
||||
request=original_request,
|
||||
)
|
||||
|
||||
|
||||
def _apply_model_prefix(data: dict) -> dict: # mutable-ok: request body handed to the dict-typed base transform_request
|
||||
"""Restore the ``modelType/model`` qualifier on the outbound model id.
|
||||
|
||||
Only prefix ids that lost their qualifier, so a cross-provider id
|
||||
(``clinepass/openrouter/foo`` -> ``openrouter/foo``) is forwarded unchanged.
|
||||
"""
|
||||
model = data.get("model")
|
||||
if isinstance(model, str) and "/" not in model:
|
||||
data["model"] = f"{CLINEPASS_MODEL_PREFIX}{model}"
|
||||
return data
|
||||
|
||||
|
||||
class ClinePassConfig(OpenAIGPTConfig):
|
||||
"""
|
||||
ClinePass configuration, inheriting the OpenAI chat transforms.
|
||||
|
||||
Overrides only the request/response points where ClinePass diverges; see the
|
||||
module docstring for the two quirks.
|
||||
"""
|
||||
|
||||
@staticmethod
|
||||
def get_realtime_http_config(model: str) -> NoReturn:
|
||||
from litellm.exceptions import BadRequestError
|
||||
|
||||
raise BadRequestError(
|
||||
message="ClinePass does not support realtime endpoints",
|
||||
model=model,
|
||||
llm_provider="clinepass",
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def validate_moderation(model: str | None, custom_llm_provider: str | None = None) -> None:
|
||||
if custom_llm_provider != "clinepass" and not (model or "").startswith("clinepass/"):
|
||||
return
|
||||
from litellm.exceptions import BadRequestError
|
||||
|
||||
raise BadRequestError(
|
||||
message="ClinePass does not support moderation endpoints",
|
||||
model=model or "",
|
||||
llm_provider="clinepass",
|
||||
)
|
||||
|
||||
def _get_openai_compatible_provider_info(
|
||||
self, api_base: str | None, api_key: str | None
|
||||
) -> tuple[str | None, str | None]:
|
||||
api_base = api_base or get_secret_str("CLINEPASS_API_BASE") or CLINEPASS_API_BASE
|
||||
dynamic_api_key = api_key or get_secret_str("CLINEPASS_API_KEY")
|
||||
return api_base, dynamic_api_key
|
||||
|
||||
def get_complete_url(
|
||||
self,
|
||||
api_base: str | None,
|
||||
api_key: str | None,
|
||||
model: str,
|
||||
optional_params: dict, # mutable-ok: matches the dict-typed base-class signature
|
||||
litellm_params: dict, # mutable-ok: matches the dict-typed base-class signature
|
||||
stream: bool | None = None,
|
||||
) -> str:
|
||||
if not api_base:
|
||||
api_base = CLINEPASS_API_BASE
|
||||
|
||||
api_base = api_base.rstrip("/")
|
||||
if api_base.endswith("/chat/completions"):
|
||||
return api_base
|
||||
|
||||
return f"{api_base}/chat/completions"
|
||||
|
||||
def get_models(
|
||||
self, api_key: str | None = None, api_base: str | None = None
|
||||
) -> list[str]: # mutable-ok: matches the dict-typed base-class signature
|
||||
"""ClinePass exposes no model catalog.
|
||||
|
||||
``GET https://api.cline.bot/api/v1/models`` returns HTTP 404, and the
|
||||
inherited OpenAI implementation would additionally ask for it at the
|
||||
wrong path -- it rewrites the base URL down to scheme+host and appends
|
||||
``/v1/models``. Return an empty catalog rather than making a request
|
||||
that is known to fail.
|
||||
"""
|
||||
return []
|
||||
|
||||
def map_openai_params(
|
||||
self,
|
||||
non_default_params: dict, # mutable-ok: matches the dict-typed base-class signature
|
||||
optional_params: dict, # mutable-ok: matches the dict-typed base-class signature
|
||||
model: str,
|
||||
drop_params: bool,
|
||||
) -> dict: # mutable-ok: matches the dict-typed base-class signature
|
||||
"""ClinePass takes the legacy ``max_tokens`` spelling only."""
|
||||
mapped_params = super().map_openai_params(
|
||||
non_default_params=non_default_params,
|
||||
optional_params=optional_params,
|
||||
model=model,
|
||||
drop_params=drop_params,
|
||||
)
|
||||
if "max_completion_tokens" in mapped_params:
|
||||
mapped_params["max_tokens"] = mapped_params.pop("max_completion_tokens")
|
||||
return mapped_params
|
||||
|
||||
def transform_request(
|
||||
self,
|
||||
model: str,
|
||||
messages: list[AllMessageValues], # mutable-ok: matches the dict-typed base-class signature
|
||||
optional_params: dict, # mutable-ok: matches the dict-typed base-class signature
|
||||
litellm_params: dict, # mutable-ok: matches the dict-typed base-class signature
|
||||
headers: dict, # mutable-ok: matches the dict-typed base-class signature
|
||||
) -> dict: # mutable-ok: matches the dict-typed base-class signature
|
||||
# BaseLLMHTTPHandler builds the body with this synchronous method on
|
||||
# both the sync and the async path, so there is deliberately no
|
||||
# async_transform_request() override -- it would never be called.
|
||||
data = super().transform_request(
|
||||
model=model,
|
||||
messages=messages,
|
||||
optional_params=optional_params,
|
||||
litellm_params=litellm_params,
|
||||
headers=headers,
|
||||
)
|
||||
return _apply_model_prefix(data)
|
||||
|
||||
def transform_response(
|
||||
self,
|
||||
model: str,
|
||||
raw_response: httpx.Response,
|
||||
model_response: ModelResponse,
|
||||
logging_obj: LiteLLMLoggingObj,
|
||||
request_data: dict, # mutable-ok: matches the dict-typed base-class signature
|
||||
messages: list[AllMessageValues], # mutable-ok: matches the dict-typed base-class signature
|
||||
optional_params: dict, # mutable-ok: matches the dict-typed base-class signature
|
||||
litellm_params: dict, # mutable-ok: matches the dict-typed base-class signature
|
||||
encoding: "Tokenizer | None",
|
||||
api_key: str | None = None,
|
||||
json_mode: bool | None = None,
|
||||
) -> ModelResponse:
|
||||
# ClinePass was once observed returning finish_reason "stop" on a completion
|
||||
# cut off by max_tokens. Follow-up probes on 2026-08-22 did not reproduce it.
|
||||
# The provider therefore reports the upstream finish reason unmodified: inferring
|
||||
# truncation from usage equalling the cap produces false positives on natural
|
||||
# completions that happen to land exactly on the cap.
|
||||
return super().transform_response(
|
||||
model=model,
|
||||
raw_response=_unwrap_response_envelope(raw_response),
|
||||
model_response=model_response,
|
||||
logging_obj=logging_obj,
|
||||
request_data=request_data,
|
||||
messages=messages,
|
||||
optional_params=optional_params,
|
||||
litellm_params=litellm_params,
|
||||
encoding=encoding,
|
||||
api_key=api_key,
|
||||
json_mode=json_mode,
|
||||
)
|
||||
|
||||
def get_error_class(
|
||||
self,
|
||||
error_message: str,
|
||||
status_code: int,
|
||||
headers: dict | httpx.Headers, # mutable-ok: matches the dict-typed base-class signature
|
||||
) -> BaseLLMException:
|
||||
return ClinePassException(
|
||||
message=error_message,
|
||||
status_code=status_code,
|
||||
headers=headers,
|
||||
)
|
||||
5
litellm/llms/clinepass/common_utils.py
Normal file
5
litellm/llms/clinepass/common_utils.py
Normal file
|
|
@ -0,0 +1,5 @@
|
|||
from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
||||
|
||||
|
||||
class ClinePassException(BaseLLMException):
|
||||
"""ClinePass exception handling class"""
|
||||
|
|
@ -2450,6 +2450,47 @@ def _complete_aiohttp_openai(
|
|||
)
|
||||
|
||||
|
||||
def _complete_http_provider(ctx: _CompletionDispatchContext) -> _CompletionDispatchResult:
|
||||
acompletion: Final = ctx.acompletion
|
||||
api_base: Final = ctx.api_base
|
||||
api_key: Final = ctx.api_key
|
||||
client: Final = _dispatch_client_http(ctx)
|
||||
custom_llm_provider: Final = ctx.custom_llm_provider
|
||||
headers: Final = ctx.headers
|
||||
litellm_params: Final = ctx.litellm_params
|
||||
logging: Final = ctx.logging
|
||||
messages: Final = ctx.messages
|
||||
model: Final = ctx.model
|
||||
model_response: Final = ctx.model_response
|
||||
optional_params: Final = ctx.optional_params
|
||||
provider_config: Final = ctx.provider_config
|
||||
shared_session: Final = ctx.shared_session
|
||||
stream: Final = ctx.stream
|
||||
timeout: Final = ctx.timeout
|
||||
|
||||
response: Final = base_llm_http_handler.completion(
|
||||
model=model,
|
||||
messages=messages, # pyright: ignore[reportUnknownArgumentType] # ctx.messages is list[Unknown]
|
||||
headers=headers, # pyright: ignore[reportUnknownArgumentType] # ctx.headers is dict[Unknown, Unknown]
|
||||
model_response=model_response,
|
||||
api_key=api_key,
|
||||
api_base=api_base,
|
||||
acompletion=acompletion,
|
||||
logging_obj=logging,
|
||||
optional_params=optional_params,
|
||||
litellm_params=litellm_params,
|
||||
shared_session=shared_session,
|
||||
timeout=timeout,
|
||||
client=client,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
encoding=_get_encoding(),
|
||||
stream=stream,
|
||||
provider_config=provider_config,
|
||||
)
|
||||
|
||||
return response
|
||||
|
||||
|
||||
def _complete_cometapi(ctx: _CompletionDispatchContext) -> _CompletionDispatchResult:
|
||||
acompletion: Final = ctx.acompletion
|
||||
api_base = ctx.api_base
|
||||
|
|
@ -5928,6 +5969,8 @@ def completion(
|
|||
response = _complete_aiohttp_openai(_dispatch_ctx)
|
||||
elif custom_llm_provider == "cometapi":
|
||||
response = _complete_cometapi(_dispatch_ctx)
|
||||
elif custom_llm_provider == "clinepass":
|
||||
response = _complete_http_provider(_dispatch_ctx)
|
||||
elif custom_llm_provider == "minimax":
|
||||
response = _complete_minimax(_dispatch_ctx)
|
||||
elif custom_llm_provider == "hosted_vllm":
|
||||
|
|
@ -7775,6 +7818,10 @@ def adapter_completion(*, adapter_id: str, **kwargs) -> BaseModel | AdapterCompl
|
|||
|
||||
|
||||
def moderation(input: str, model: str | None = None, api_key: str | None = None, **kwargs) -> OpenAIModerationResponse:
|
||||
custom_llm_provider: Final[object] = kwargs.get("custom_llm_provider")
|
||||
litellm.ClinePassConfig.validate_moderation(
|
||||
model=model, custom_llm_provider=custom_llm_provider if isinstance(custom_llm_provider, str) else None
|
||||
)
|
||||
# only supports open ai for now
|
||||
api_key = api_key or litellm.api_key or litellm.openai_key or get_secret_str("OPENAI_API_KEY")
|
||||
|
||||
|
|
@ -7809,6 +7856,7 @@ async def amoderation(
|
|||
) -> OpenAIModerationResponse:
|
||||
from openai import AsyncOpenAI
|
||||
|
||||
litellm.ClinePassConfig.validate_moderation(model=model, custom_llm_provider=custom_llm_provider)
|
||||
# only supports open ai for now
|
||||
api_key = api_key or litellm.api_key or litellm.openai_key or get_secret_str("OPENAI_API_KEY")
|
||||
optional_params: Final = GenericLiteLLMParams.model_validate(kwargs)
|
||||
|
|
|
|||
|
|
@ -15614,6 +15614,10 @@
|
|||
"prompt_cache_min_tokens": 1024,
|
||||
"source": "https://platform.claude.com/docs/en/about-claude/pricing"
|
||||
},
|
||||
"clinepass/deepseek-v4-flash": {
|
||||
"litellm_provider": "clinepass",
|
||||
"mode": "chat"
|
||||
},
|
||||
"cloudflare/clef": {
|
||||
"input_cost_per_token": 2.4e-07,
|
||||
"litellm_provider": "cloudflare",
|
||||
|
|
|
|||
|
|
@ -492,6 +492,24 @@
|
|||
"interactions": true
|
||||
}
|
||||
},
|
||||
"clinepass": {
|
||||
"display_name": "ClinePass (`clinepass`)",
|
||||
"url": "https://docs.litellm.ai/docs/providers/clinepass",
|
||||
"endpoints": {
|
||||
"chat_completions": true,
|
||||
"messages": true,
|
||||
"responses": true,
|
||||
"embeddings": false,
|
||||
"image_generations": false,
|
||||
"audio_transcriptions": false,
|
||||
"audio_speech": false,
|
||||
"moderations": false,
|
||||
"batches": false,
|
||||
"rerank": false,
|
||||
"a2a": true,
|
||||
"interactions": true
|
||||
}
|
||||
},
|
||||
"cloudflare": {
|
||||
"display_name": "Cloudflare AI Workers (`cloudflare`)",
|
||||
"url": "https://docs.litellm.ai/docs/providers/cloudflare_workers",
|
||||
|
|
|
|||
|
|
@ -829,6 +829,34 @@
|
|||
],
|
||||
"default_model_placeholder": "gpt-3.5-turbo"
|
||||
},
|
||||
{
|
||||
"provider": "CLINEPASS",
|
||||
"provider_display_name": "ClinePass",
|
||||
"litellm_provider": "clinepass",
|
||||
"credential_fields": [
|
||||
{
|
||||
"key": "api_base",
|
||||
"label": "API Base",
|
||||
"placeholder": null,
|
||||
"tooltip": null,
|
||||
"required": false,
|
||||
"field_type": "text",
|
||||
"options": null,
|
||||
"default_value": null
|
||||
},
|
||||
{
|
||||
"key": "api_key",
|
||||
"label": "API Key",
|
||||
"placeholder": null,
|
||||
"tooltip": null,
|
||||
"required": false,
|
||||
"field_type": "password",
|
||||
"options": null,
|
||||
"default_value": null
|
||||
}
|
||||
],
|
||||
"default_model_placeholder": "gpt-3.5-turbo"
|
||||
},
|
||||
{
|
||||
"provider": "CLOUDFLARE",
|
||||
"provider_display_name": "Cloudflare",
|
||||
|
|
|
|||
|
|
@ -2408,10 +2408,13 @@ async def _aresponses_websocket(
|
|||
|
||||
resolved_api_key: Final = (
|
||||
dynamic_api_key
|
||||
or api_key
|
||||
or litellm_params.api_key
|
||||
or litellm.api_key
|
||||
or litellm.openai_key
|
||||
or get_secret_str("OPENAI_API_KEY")
|
||||
or (
|
||||
(litellm.api_key or litellm.openai_key or get_secret_str("OPENAI_API_KEY"))
|
||||
if responses_api_provider_config is not None
|
||||
else None
|
||||
)
|
||||
)
|
||||
|
||||
# Extract params that we're passing explicitly to avoid duplicates in **kwargs
|
||||
|
|
|
|||
|
|
@ -2824,9 +2824,10 @@ class ManagedResponsesWebSocketHandler:
|
|||
|
||||
def _inject_credentials(self, call_kwargs: dict[str, object], model: str | None = None) -> None:
|
||||
"""Inject connection-level credentials and metadata into call_kwargs."""
|
||||
if self.api_key is not None:
|
||||
same_provider: Final = self._same_provider(model)
|
||||
if self.api_key is not None and same_provider:
|
||||
call_kwargs["api_key"] = self.api_key
|
||||
if self.api_base is not None:
|
||||
if self.api_base is not None and same_provider:
|
||||
call_kwargs["api_base"] = self.api_base
|
||||
if self.timeout is not None:
|
||||
call_kwargs["timeout"] = self.timeout
|
||||
|
|
@ -2835,7 +2836,7 @@ class ManagedResponsesWebSocketHandler:
|
|||
# (e.g., connection is vertex_ai but event says openai/gpt-4), let litellm
|
||||
# re-resolve from the model string. Same-provider model variants (e.g.,
|
||||
# vertex_ai/gemini-2.0 -> vertex_ai/gemini-1.5) still inherit the provider.
|
||||
if self.custom_llm_provider is not None and self._same_provider(model):
|
||||
if self.custom_llm_provider is not None and same_provider:
|
||||
call_kwargs["custom_llm_provider"] = self.custom_llm_provider
|
||||
if self.litellm_metadata:
|
||||
call_kwargs["litellm_metadata"] = dict(self.litellm_metadata)
|
||||
|
|
@ -2960,11 +2961,18 @@ class ManagedResponsesWebSocketHandler:
|
|||
call_kwargs: Final = self._build_base_call_kwargs(msg_obj)
|
||||
call_kwargs["stream"] = True
|
||||
|
||||
# A frame that repeats the connection's public alias (model_group) must
|
||||
# reuse the router-resolved self.model; passing the alias raw to
|
||||
# litellm.aresponses fails in get_llm_provider. A genuinely different
|
||||
# provider-prefixed per-frame model is still honored.
|
||||
requested_model: Final[str | None] = _optional_str(call_kwargs.pop("model", None))
|
||||
authorized_models: Final = (self.model, self.model_group, f"{self.custom_llm_provider}/{self.model}")
|
||||
if (
|
||||
self.user_api_key_dict is not None
|
||||
and requested_model is not None
|
||||
and requested_model not in authorized_models
|
||||
):
|
||||
await self._send_error(
|
||||
"Changing models requires a new authorized WebSocket connection",
|
||||
error_type="invalid_request_error",
|
||||
)
|
||||
return
|
||||
model: Final[str] = (
|
||||
self.model if requested_model is None or requested_model == self.model_group else requested_model
|
||||
)
|
||||
|
|
|
|||
|
|
@ -4166,6 +4166,7 @@ class LlmProviders(str, Enum):
|
|||
APERTIS = "apertis"
|
||||
NANOGPT = "nano-gpt"
|
||||
POE = "poe"
|
||||
CLINEPASS = "clinepass"
|
||||
CHUTES = "chutes"
|
||||
NEOSANTARA = "neosantara"
|
||||
PARASAIL = "parasail"
|
||||
|
|
|
|||
|
|
@ -8504,6 +8504,7 @@ class ProviderConfigManager:
|
|||
LlmProviders.EDENAI: (litellm.EdenAIChatConfig, False),
|
||||
LlmProviders.FAL_AI: (litellm.FalAIChatConfig, False),
|
||||
LlmProviders.COMETAPI: (lambda: litellm.CometAPIConfig(), False),
|
||||
LlmProviders.CLINEPASS: (litellm.ClinePassConfig, False),
|
||||
LlmProviders.DATAROBOT: (lambda: litellm.DataRobotConfig(), False),
|
||||
LlmProviders.GEMINI: (lambda: litellm.GoogleAIStudioGeminiConfig(), False),
|
||||
LlmProviders.AI21: (lambda: litellm.AI21ChatConfig(), False),
|
||||
|
|
@ -9716,6 +9717,8 @@ class ProviderConfigManager:
|
|||
(POST /realtime/client_secrets and POST /realtime/calls).
|
||||
"""
|
||||
|
||||
if LlmProviders.CLINEPASS == provider:
|
||||
return litellm.ClinePassConfig.get_realtime_http_config(model=model)
|
||||
if LlmProviders.OPENAI == provider:
|
||||
from litellm.llms.openai.realtime.http_transformation import (
|
||||
OpenAIRealtimeHTTPConfig,
|
||||
|
|
|
|||
|
|
@ -15614,6 +15614,10 @@
|
|||
"prompt_cache_min_tokens": 1024,
|
||||
"source": "https://platform.claude.com/docs/en/about-claude/pricing"
|
||||
},
|
||||
"clinepass/deepseek-v4-flash": {
|
||||
"litellm_provider": "clinepass",
|
||||
"mode": "chat"
|
||||
},
|
||||
"cloudflare/clef": {
|
||||
"input_cost_per_token": 2.4e-07,
|
||||
"litellm_provider": "cloudflare",
|
||||
|
|
|
|||
|
|
@ -546,6 +546,24 @@
|
|||
"interactions": true
|
||||
}
|
||||
},
|
||||
"clinepass": {
|
||||
"display_name": "ClinePass (`clinepass`)",
|
||||
"url": "https://docs.litellm.ai/docs/providers/clinepass",
|
||||
"endpoints": {
|
||||
"chat_completions": true,
|
||||
"messages": true,
|
||||
"responses": true,
|
||||
"embeddings": false,
|
||||
"image_generations": false,
|
||||
"audio_transcriptions": false,
|
||||
"audio_speech": false,
|
||||
"moderations": false,
|
||||
"batches": false,
|
||||
"rerank": false,
|
||||
"a2a": true,
|
||||
"interactions": true
|
||||
}
|
||||
},
|
||||
"cloudflare": {
|
||||
"display_name": "Cloudflare AI Workers (`cloudflare`)",
|
||||
"url": "https://docs.litellm.ai/docs/providers/cloudflare_workers",
|
||||
|
|
|
|||
|
|
@ -812,6 +812,7 @@ PROVIDERS_WITH_A_HANDLER = (
|
|||
"azure",
|
||||
"azure_ai",
|
||||
"bedrock",
|
||||
"clinepass",
|
||||
"cloudflare",
|
||||
"cohere",
|
||||
"databricks",
|
||||
|
|
@ -1048,6 +1049,7 @@ PROVIDERS_THAT_RECOGNISE_A_FULL_CONTEXT_WINDOW = (
|
|||
"anthropic",
|
||||
"azure",
|
||||
"azure_ai",
|
||||
"clinepass",
|
||||
"databricks",
|
||||
"deepseek",
|
||||
"fireworks_ai",
|
||||
|
|
@ -1066,6 +1068,7 @@ PROVIDERS_THAT_RECOGNISE_A_CONTENT_POLICY_BLOCK = (
|
|||
"ai21",
|
||||
"azure",
|
||||
"azure_ai",
|
||||
"clinepass",
|
||||
"deepseek",
|
||||
"fireworks_ai",
|
||||
"groq",
|
||||
|
|
|
|||
0
tests/unit/llms/clinepass/__init__.py
Normal file
0
tests/unit/llms/clinepass/__init__.py
Normal file
0
tests/unit/llms/clinepass/chat/__init__.py
Normal file
0
tests/unit/llms/clinepass/chat/__init__.py
Normal file
|
|
@ -0,0 +1,925 @@
|
|||
"""Tests for the ClinePass provider.
|
||||
|
||||
The point of the end-to-end tests here is that they drive ``litellm.completion()``
|
||||
with a mocked transport rather than calling the transforms directly -- a unit test
|
||||
that calls ``transform_response()`` itself proves the function is correct but not
|
||||
that anything invokes it, which is exactly how the envelope unwrap was previously
|
||||
shipped as dead code.
|
||||
"""
|
||||
|
||||
import json
|
||||
from typing import Final
|
||||
from unittest.mock import patch
|
||||
|
||||
import httpx
|
||||
import pytest
|
||||
from starlette.websockets import WebSocket
|
||||
|
||||
import litellm
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging
|
||||
from litellm.llms.clinepass.chat.transformation import (
|
||||
ClinePassConfig,
|
||||
_apply_model_prefix,
|
||||
_unwrap_response_envelope,
|
||||
)
|
||||
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
from litellm.responses.main import _aresponses_websocket
|
||||
from litellm.types.utils import LlmProviders
|
||||
from litellm.utils import ProviderConfigManager
|
||||
|
||||
API_KEY = "sk-clinepass-test-not-real"
|
||||
|
||||
ENVELOPED_COMPLETION = {
|
||||
"success": True,
|
||||
"data": {
|
||||
"id": "chatcmpl-test",
|
||||
"object": "chat.completion",
|
||||
"created": 1,
|
||||
"model": "clinepass/deepseek-v4-flash",
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"finish_reason": "stop",
|
||||
"message": {
|
||||
"role": "assistant",
|
||||
"content": "pong",
|
||||
"reasoning": "the user asked for pong",
|
||||
},
|
||||
}
|
||||
],
|
||||
"usage": {"prompt_tokens": 5, "completion_tokens": 2, "total_tokens": 7},
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def _response(payload: dict, url: str = "https://api.cline.bot/api/v1/chat/completions") -> httpx.Response:
|
||||
return httpx.Response(200, json=payload, request=httpx.Request("POST", url))
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _clinepass_env(monkeypatch):
|
||||
monkeypatch.setenv("CLINEPASS_API_KEY", API_KEY)
|
||||
monkeypatch.delenv("CLINEPASS_API_BASE", raising=False)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def unrelated_credentials(monkeypatch):
|
||||
for env_name in ("OPENAI_API_KEY", "ANTHROPIC_API_KEY", "GROQ_API_KEY", "OPENROUTER_API_KEY"):
|
||||
monkeypatch.setenv(env_name, f"sk-unrelated-{env_name}")
|
||||
for attribute in ("api_key", "openai_key", "anthropic_key"):
|
||||
monkeypatch.setattr(litellm, attribute, f"sk-unrelated-{attribute}")
|
||||
|
||||
|
||||
@pytest.mark.parametrize("credential_source", ["missing", "environment", "explicit"])
|
||||
@pytest.mark.parametrize("custom_llm_provider", [None, "clinepass"])
|
||||
def test_chat_sends_only_clinepass_credentials(
|
||||
monkeypatch, unrelated_credentials, credential_source, custom_llm_provider
|
||||
):
|
||||
if credential_source == "missing":
|
||||
monkeypatch.delenv("CLINEPASS_API_KEY", raising=False)
|
||||
explicit_key: Final = "cp-request-key" if credential_source == "explicit" else None
|
||||
captured = {}
|
||||
|
||||
def fake_post(self, url, *args, **kwargs):
|
||||
captured["url"] = str(url)
|
||||
captured["headers"] = httpx.Headers(kwargs["headers"])
|
||||
return _response(ENVELOPED_COMPLETION)
|
||||
|
||||
with patch.object(HTTPHandler, "post", fake_post):
|
||||
response = litellm.completion(
|
||||
model="deepseek-v4-flash" if custom_llm_provider else "clinepass/deepseek-v4-flash",
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
api_key=explicit_key,
|
||||
messages=[{"role": "user", "content": "ping"}],
|
||||
)
|
||||
|
||||
expected_key: Final = explicit_key or (API_KEY if credential_source == "environment" else None)
|
||||
assert captured["url"] == "https://api.cline.bot/api/v1/chat/completions"
|
||||
assert captured["headers"].get("authorization") == (f"Bearer {expected_key}" if expected_key else None)
|
||||
assert response.choices[0].message.content == "pong"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("credential_source", ["missing", "environment", "explicit"])
|
||||
@pytest.mark.asyncio
|
||||
async def test_async_chat_sends_only_clinepass_credentials(monkeypatch, unrelated_credentials, credential_source):
|
||||
if credential_source == "missing":
|
||||
monkeypatch.delenv("CLINEPASS_API_KEY", raising=False)
|
||||
explicit_key: Final = "cp-request-key" if credential_source == "explicit" else None
|
||||
captured = {}
|
||||
|
||||
async def fake_post(self, url, *args, **kwargs):
|
||||
captured["url"] = str(url)
|
||||
captured["headers"] = httpx.Headers(kwargs["headers"])
|
||||
return _response(ENVELOPED_COMPLETION)
|
||||
|
||||
with patch.object(AsyncHTTPHandler, "post", fake_post):
|
||||
response = await litellm.acompletion(
|
||||
model="clinepass/deepseek-v4-flash",
|
||||
api_key=explicit_key,
|
||||
messages=[{"role": "user", "content": "ping"}],
|
||||
)
|
||||
|
||||
expected_key: Final = explicit_key or (API_KEY if credential_source == "environment" else None)
|
||||
assert captured["url"] == "https://api.cline.bot/api/v1/chat/completions"
|
||||
assert captured["headers"].get("authorization") == (f"Bearer {expected_key}" if expected_key else None)
|
||||
assert response.choices[0].message.content == "pong"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("credential_source", ["missing", "environment", "explicit"])
|
||||
@pytest.mark.parametrize("connection_provider", ["clinepass", "mistral"])
|
||||
@pytest.mark.parametrize("changed_model", [None, "openai/gpt-4o", "clinepass/unauthorized-model"])
|
||||
@pytest.mark.asyncio
|
||||
async def test_managed_responses_websocket_sends_only_clinepass_credentials(
|
||||
monkeypatch, unrelated_credentials, credential_source, connection_provider, changed_model
|
||||
):
|
||||
if credential_source == "missing":
|
||||
monkeypatch.delenv("CLINEPASS_API_KEY", raising=False)
|
||||
connection_key: Final = (
|
||||
"sk-unrelated-mistral"
|
||||
if connection_provider == "mistral"
|
||||
else "cp-request-key"
|
||||
if credential_source == "explicit"
|
||||
else None
|
||||
)
|
||||
sent = []
|
||||
received = []
|
||||
foreign_requests = []
|
||||
lifecycle = iter(
|
||||
(
|
||||
{"type": "websocket.connect"},
|
||||
*(
|
||||
(
|
||||
{
|
||||
"type": "websocket.receive",
|
||||
"text": json.dumps({"type": "response.create", "model": changed_model, "input": "hi"}),
|
||||
},
|
||||
)
|
||||
if changed_model is not None
|
||||
else ()
|
||||
),
|
||||
{"type": "websocket.disconnect", "code": 1000},
|
||||
)
|
||||
)
|
||||
|
||||
async def block_foreign_request(self, request, *args, **kwargs):
|
||||
foreign_requests.append(str(request.url))
|
||||
raise AssertionError("Unexpected provider HTTP request")
|
||||
|
||||
monkeypatch.setattr(httpx.AsyncClient, "send", block_foreign_request)
|
||||
|
||||
async def receive():
|
||||
return next(lifecycle)
|
||||
|
||||
async def send(message):
|
||||
if message["type"] == "websocket.send":
|
||||
received.append(json.loads(message["text"]))
|
||||
|
||||
websocket: Final = WebSocket(
|
||||
scope={"type": "websocket", "path": "/v1/responses", "headers": [], "query_string": b""},
|
||||
receive=receive,
|
||||
send=send,
|
||||
)
|
||||
await websocket.accept()
|
||||
chunk: Final = {
|
||||
"id": "chatcmpl-test",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 1,
|
||||
"model": "cline-pass/deepseek-v4-flash",
|
||||
"choices": [{"index": 0, "delta": {"role": "assistant", "content": "pong"}, "finish_reason": "stop"}],
|
||||
}
|
||||
|
||||
async def fake_post(self, url, *args, **kwargs):
|
||||
sent.append((str(url), httpx.Headers(kwargs["headers"]).get("authorization")))
|
||||
return httpx.Response(
|
||||
200,
|
||||
content=f"data: {json.dumps(chunk)}\n\ndata: [DONE]\n\n".encode(),
|
||||
headers={"content-type": "text/event-stream"},
|
||||
request=httpx.Request("POST", str(url)),
|
||||
)
|
||||
|
||||
with patch.object(AsyncHTTPHandler, "post", fake_post):
|
||||
result = await _aresponses_websocket.__wrapped__(
|
||||
model=f"{connection_provider}/deepseek-v4-flash",
|
||||
websocket=websocket,
|
||||
api_key=connection_key,
|
||||
user_api_key_dict=(
|
||||
UserAPIKeyAuth(models=[f"{connection_provider}/deepseek-v4-flash"])
|
||||
if changed_model is not None
|
||||
else None
|
||||
),
|
||||
first_message=json.dumps(
|
||||
{"type": "response.create", "model": "clinepass/deepseek-v4-flash", "input": "ping"}
|
||||
),
|
||||
litellm_logging_obj=Logging(
|
||||
model=f"{connection_provider}/deepseek-v4-flash",
|
||||
messages=[],
|
||||
stream=True,
|
||||
call_type="aresponses",
|
||||
start_time=0,
|
||||
litellm_call_id="cp-ws-test",
|
||||
function_id="cp-ws-test",
|
||||
),
|
||||
)
|
||||
|
||||
expected_key: Final = (
|
||||
connection_key
|
||||
if connection_provider == "clinepass" and credential_source == "explicit"
|
||||
else API_KEY
|
||||
if credential_source != "missing"
|
||||
else None
|
||||
)
|
||||
assert sent == (
|
||||
[]
|
||||
if changed_model is not None and connection_provider == "mistral"
|
||||
else [("https://api.cline.bot/api/v1/chat/completions", f"Bearer {expected_key}" if expected_key else None)]
|
||||
)
|
||||
assert result is None
|
||||
errors: Final = [event["error"] for event in received if event["type"] == "error"]
|
||||
assert errors == (
|
||||
[
|
||||
{
|
||||
"type": "invalid_request_error",
|
||||
"message": "Changing models requires a new authorized WebSocket connection",
|
||||
}
|
||||
]
|
||||
* (2 if connection_provider == "mistral" else 1)
|
||||
if changed_model is not None
|
||||
else []
|
||||
)
|
||||
assert foreign_requests == []
|
||||
assert ("response.completed" in [event["type"] for event in received]) == bool(sent)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Registration / routing
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_get_llm_provider_resolves_clinepass():
|
||||
model, provider, api_key, api_base = litellm.get_llm_provider(model="clinepass/deepseek-v4-flash")
|
||||
assert model == "deepseek-v4-flash"
|
||||
assert provider == "clinepass"
|
||||
assert api_key == API_KEY
|
||||
assert api_base == "https://api.cline.bot/api/v1"
|
||||
|
||||
|
||||
def test_provider_config_manager_returns_clinepass_config():
|
||||
config = ProviderConfigManager.get_provider_chat_config(model="deepseek-v4-flash", provider=LlmProviders.CLINEPASS)
|
||||
assert isinstance(config, ClinePassConfig)
|
||||
|
||||
|
||||
def test_clinepass_is_not_a_json_configured_provider_via_behaviour():
|
||||
"""ClinePass needs a response transform, which the JSON provider system's
|
||||
OpenAI-SDK dispatch path never invokes. We assert it is not on that path
|
||||
by verifying the envelope unwrap actually triggers."""
|
||||
captured = {}
|
||||
|
||||
def fake_post(self, url, *args, **kwargs):
|
||||
captured["body"] = json.loads(kwargs["data"])
|
||||
return _response(ENVELOPED_COMPLETION)
|
||||
|
||||
with patch.object(HTTPHandler, "post", fake_post):
|
||||
response = litellm.completion(
|
||||
model="clinepass/deepseek-v4-flash",
|
||||
messages=[{"role": "user", "content": "ping"}],
|
||||
)
|
||||
|
||||
# The unwrap works, meaning we didn't drift into the JSON provider registry
|
||||
# which would have bypassed our custom transform.
|
||||
assert response.choices[0].message.content == "pong"
|
||||
|
||||
|
||||
def test_clinepass_is_not_in_openai_compatible_providers():
|
||||
"""The cheap structural guard for the credential leak. Keep it.
|
||||
|
||||
The behavioural test below proves the *consequence*; this proves the
|
||||
*cause*, in one line and with no mocking that could itself be wrong. Both
|
||||
are wanted: a mocked behavioural test can drift into passing for the wrong
|
||||
reason, while this cannot.
|
||||
|
||||
The membership is not routing-inert, which is what made it dangerous. The
|
||||
list is also read by the speech branch in `main.py` -- which sends to the
|
||||
provider's own `api_base` while taking the key from `OPENAI_API_KEY`, so a
|
||||
`litellm.speech(model="clinepass/...")` call shipped the caller's OpenAI
|
||||
credential to the Cline host for an endpoint ClinePass does not implement --
|
||||
by `OPENAI_AUDIO_TRANSCRIPTION_PROVIDERS`, which is derived from it, by
|
||||
image generation in `litellm/images/main.py`, and by
|
||||
`_add_provider_specific_params`, which nests unknown kwargs under
|
||||
`extra_body` for listed providers. `BaseLLMHTTPHandler` merges that back into
|
||||
the request body, so the wire body is the same either way: membership is
|
||||
pinned structurally here and at the params level in
|
||||
`test_unknown_kwargs_are_flattened_not_wrapped_in_extra_body`, not by the
|
||||
request body.
|
||||
|
||||
Exception mapping is preserved by registering ClinePass explicitly beside
|
||||
`mistral` in `exception_mapping_utils.py`; see
|
||||
`test_upstream_401_maps_to_authentication_error`.
|
||||
"""
|
||||
assert "clinepass" not in litellm.openai_compatible_providers
|
||||
assert "clinepass" not in litellm.constants.OPENAI_AUDIO_TRANSCRIPTION_PROVIDERS
|
||||
|
||||
|
||||
def test_clinepass_is_not_in_openai_compatible_providers_via_behaviour():
|
||||
"""ClinePass must NOT be in `openai_compatible_providers`.
|
||||
The drift guard here is the transcription half: a listed provider is picked up
|
||||
by the OpenAI transcription branch instead of raising as unmapped.
|
||||
The body assertions only pin the contract that an unknown kwarg reaches the
|
||||
JSON body flat; they hold for listed providers too, so they do not detect
|
||||
drift (the request body is identical either way)."""
|
||||
captured = {}
|
||||
|
||||
def fake_post(self, url, *args, **kwargs):
|
||||
captured["body"] = json.loads(kwargs["data"])
|
||||
return _response(ENVELOPED_COMPLETION)
|
||||
|
||||
with patch.object(HTTPHandler, "post", fake_post):
|
||||
litellm.completion(
|
||||
model="clinepass/deepseek-v4-flash",
|
||||
messages=[{"role": "user", "content": "ping"}],
|
||||
custom_vendor_flag=True, # unknown kwarg
|
||||
)
|
||||
|
||||
assert "extra_body" not in captured["body"]
|
||||
assert captured["body"].get("custom_vendor_flag") is True
|
||||
|
||||
# Audio transcription should outright fail as unmapped, confirming it
|
||||
# isn't implicitly picked up by OPENAI_AUDIO_TRANSCRIPTION_PROVIDERS
|
||||
with pytest.raises(ValueError, match="Unmapped provider"):
|
||||
litellm.transcription(
|
||||
model="clinepass/deepseek-v4-flash",
|
||||
file=b"fake audio data",
|
||||
)
|
||||
|
||||
|
||||
def test_api_base_env_override(monkeypatch):
|
||||
monkeypatch.setenv("CLINEPASS_API_BASE", "https://proxy.internal/api/v1")
|
||||
_, _, _, api_base = litellm.get_llm_provider(model="clinepass/deepseek-v4-flash")
|
||||
assert api_base == "https://proxy.internal/api/v1"
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"api_base,expected",
|
||||
[
|
||||
(None, "https://api.cline.bot/api/v1/chat/completions"),
|
||||
("https://api.cline.bot/api/v1", "https://api.cline.bot/api/v1/chat/completions"),
|
||||
("https://api.cline.bot/api/v1/", "https://api.cline.bot/api/v1/chat/completions"),
|
||||
(
|
||||
"https://api.cline.bot/api/v1/chat/completions",
|
||||
"https://api.cline.bot/api/v1/chat/completions",
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_get_complete_url(api_base, expected):
|
||||
url = ClinePassConfig().get_complete_url(
|
||||
api_base=api_base, api_key=API_KEY, model="deepseek-v4-flash", optional_params={}, litellm_params={}
|
||||
)
|
||||
assert url == expected
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Model prefix
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_model_prefix_restored_on_bare_id():
|
||||
"""The restored qualifier is the catalog namespace ``cline-pass/`` (hyphenated),
|
||||
NOT LiteLLM's own ``clinepass/`` routing prefix.
|
||||
|
||||
The API validates only the *shape* of a model id, so a wrong namespace still
|
||||
returns HTTP 200 -- but it does not always resolve to the same underlying
|
||||
model, which makes a wrong value silent rather than harmless.
|
||||
"""
|
||||
assert _apply_model_prefix({"model": "deepseek-v4-flash"})["model"] == "cline-pass/deepseek-v4-flash"
|
||||
|
||||
|
||||
def test_model_prefix_left_alone_when_qualifier_present():
|
||||
"""`clinepass/openrouter/foo` arrives here as `openrouter/foo` and must pass through."""
|
||||
assert _apply_model_prefix({"model": "openrouter/foo"})["model"] == "openrouter/foo"
|
||||
|
||||
|
||||
def test_model_prefix_ignores_missing_model():
|
||||
assert _apply_model_prefix({}) == {}
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Response envelope
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_unwrap_envelope_extracts_inner_completion():
|
||||
unwrapped = _unwrap_response_envelope(_response(ENVELOPED_COMPLETION))
|
||||
assert unwrapped.json() == ENVELOPED_COMPLETION["data"]
|
||||
|
||||
|
||||
def test_unwrap_envelope_content_length_describes_the_new_body():
|
||||
"""The original content-length describes the enveloped bytes and must not be
|
||||
carried over; httpx recomputes a correct one for the rewritten body."""
|
||||
raw = _response(ENVELOPED_COMPLETION)
|
||||
unwrapped = _unwrap_response_envelope(raw)
|
||||
assert unwrapped.headers["content-length"] != raw.headers["content-length"]
|
||||
assert int(unwrapped.headers["content-length"]) == len(unwrapped.content)
|
||||
|
||||
|
||||
def test_unwrap_envelope_passes_through_openai_shaped_body():
|
||||
payload = ENVELOPED_COMPLETION["data"]
|
||||
assert _unwrap_response_envelope(_response(payload)).json() == payload
|
||||
|
||||
|
||||
def test_unwrap_envelope_passes_through_error_nested_under_same_key():
|
||||
"""An error under `data` has no `choices` and must not be mistaken for a completion."""
|
||||
payload = {"success": False, "data": {"message": "bad model"}}
|
||||
assert _unwrap_response_envelope(_response(payload)).json() == payload
|
||||
|
||||
|
||||
def test_unwrap_envelope_passes_through_non_json_body():
|
||||
raw = httpx.Response(
|
||||
200, content=b"not json", request=httpx.Request("POST", "https://api.cline.bot/api/v1/chat/completions")
|
||||
)
|
||||
assert _unwrap_response_envelope(raw) is raw
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Parameter mapping
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_max_completion_tokens_mapped_to_max_tokens():
|
||||
mapped = ClinePassConfig().map_openai_params(
|
||||
non_default_params={"max_completion_tokens": 4000},
|
||||
optional_params={},
|
||||
model="deepseek-v4-flash",
|
||||
drop_params=False,
|
||||
)
|
||||
assert mapped == {"max_tokens": 4000}
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# End-to-end through litellm.completion() -- these are the load-bearing ones
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_completion_unwraps_envelope_and_prefixes_model():
|
||||
captured = {}
|
||||
|
||||
def fake_post(self, url, *args, **kwargs):
|
||||
captured["url"] = str(url)
|
||||
captured["body"] = json.loads(kwargs["data"])
|
||||
return _response(ENVELOPED_COMPLETION)
|
||||
|
||||
with patch.object(HTTPHandler, "post", fake_post):
|
||||
response = litellm.completion(
|
||||
model="clinepass/deepseek-v4-flash",
|
||||
messages=[{"role": "user", "content": "ping"}],
|
||||
max_tokens=4000,
|
||||
)
|
||||
|
||||
assert captured["url"] == "https://api.cline.bot/api/v1/chat/completions"
|
||||
assert captured["body"]["model"] == "cline-pass/deepseek-v4-flash"
|
||||
assert response.choices[0].message.content == "pong"
|
||||
assert response.choices[0].message.reasoning_content == "the user asked for pong"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_acompletion_unwraps_envelope_and_prefixes_model():
|
||||
captured = {}
|
||||
|
||||
async def fake_post(self, url, *args, **kwargs):
|
||||
captured["url"] = str(url)
|
||||
captured["body"] = json.loads(kwargs["data"])
|
||||
return _response(ENVELOPED_COMPLETION)
|
||||
|
||||
with patch.object(AsyncHTTPHandler, "post", fake_post):
|
||||
response = await litellm.acompletion(
|
||||
model="clinepass/deepseek-v4-flash",
|
||||
messages=[{"role": "user", "content": "ping"}],
|
||||
max_tokens=4000,
|
||||
)
|
||||
|
||||
assert captured["url"] == "https://api.cline.bot/api/v1/chat/completions"
|
||||
assert captured["body"]["model"] == "cline-pass/deepseek-v4-flash"
|
||||
assert response.choices[0].message.content == "pong"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("credential_source", ["missing", "environment", "explicit"])
|
||||
def test_completion_streaming_is_not_unwrapped(monkeypatch, unrelated_credentials, credential_source):
|
||||
"""ClinePass does NOT wrap SSE chunks -- they are already OpenAI-shaped."""
|
||||
if credential_source == "missing":
|
||||
monkeypatch.delenv("CLINEPASS_API_KEY", raising=False)
|
||||
explicit_key: Final = "cp-stream-key" if credential_source == "explicit" else None
|
||||
captured = {}
|
||||
chunks = [
|
||||
{
|
||||
"id": "chatcmpl-test",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 1,
|
||||
"model": "clinepass/deepseek-v4-flash",
|
||||
"choices": [{"index": 0, "delta": {"content": piece}, "finish_reason": None}],
|
||||
}
|
||||
for piece in ["one ", "two ", "three"]
|
||||
]
|
||||
chunks.append(
|
||||
{
|
||||
"id": "chatcmpl-test",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 1,
|
||||
"model": "clinepass/deepseek-v4-flash",
|
||||
"choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}],
|
||||
}
|
||||
)
|
||||
body = "".join(f"data: {json.dumps(c)}\n\n" for c in chunks) + "data: [DONE]\n\n"
|
||||
|
||||
def fake_post(self, url, *args, **kwargs):
|
||||
captured["authorization"] = httpx.Headers(kwargs["headers"]).get("authorization")
|
||||
return httpx.Response(
|
||||
200,
|
||||
content=body.encode(),
|
||||
headers={"content-type": "text/event-stream"},
|
||||
request=httpx.Request("POST", str(url)),
|
||||
)
|
||||
|
||||
with patch.object(HTTPHandler, "post", fake_post):
|
||||
stream = litellm.completion(
|
||||
model="clinepass/deepseek-v4-flash",
|
||||
messages=[{"role": "user", "content": "count"}],
|
||||
api_key=explicit_key,
|
||||
max_tokens=4000,
|
||||
stream=True,
|
||||
)
|
||||
text = ""
|
||||
finish_reason = None
|
||||
for c in stream:
|
||||
if c.choices:
|
||||
if c.choices[0].delta.content:
|
||||
text += c.choices[0].delta.content
|
||||
if c.choices[0].finish_reason:
|
||||
finish_reason = c.choices[0].finish_reason
|
||||
|
||||
assert text == "one two three"
|
||||
assert finish_reason == "stop"
|
||||
expected_key: Final = explicit_key or (API_KEY if credential_source == "environment" else None)
|
||||
assert captured["authorization"] == (f"Bearer {expected_key}" if expected_key else None)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("credential_source", ["missing", "environment", "explicit"])
|
||||
@pytest.mark.asyncio
|
||||
async def test_acompletion_streaming_is_not_unwrapped(monkeypatch, unrelated_credentials, credential_source):
|
||||
if credential_source == "missing":
|
||||
monkeypatch.delenv("CLINEPASS_API_KEY", raising=False)
|
||||
explicit_key: Final = "cp-stream-key" if credential_source == "explicit" else None
|
||||
captured = {}
|
||||
chunks = [
|
||||
{
|
||||
"id": "chatcmpl-test",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 1,
|
||||
"model": "clinepass/deepseek-v4-flash",
|
||||
"choices": [{"index": 0, "delta": {"content": piece}, "finish_reason": None}],
|
||||
}
|
||||
for piece in ["one ", "two ", "three"]
|
||||
]
|
||||
chunks.append(
|
||||
{
|
||||
"id": "chatcmpl-test",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 1,
|
||||
"model": "clinepass/deepseek-v4-flash",
|
||||
"choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}],
|
||||
}
|
||||
)
|
||||
body = "".join(f"data: {json.dumps(c)}\n\n" for c in chunks) + "data: [DONE]\n\n"
|
||||
|
||||
async def fake_post(self, url, *args, **kwargs):
|
||||
captured["authorization"] = httpx.Headers(kwargs["headers"]).get("authorization")
|
||||
return httpx.Response(
|
||||
200,
|
||||
content=body.encode(),
|
||||
headers={"content-type": "text/event-stream"},
|
||||
request=httpx.Request("POST", str(url)),
|
||||
)
|
||||
|
||||
with patch.object(AsyncHTTPHandler, "post", fake_post):
|
||||
stream = await litellm.acompletion(
|
||||
model="clinepass/deepseek-v4-flash",
|
||||
messages=[{"role": "user", "content": "count"}],
|
||||
api_key=explicit_key,
|
||||
max_tokens=4000,
|
||||
stream=True,
|
||||
)
|
||||
text = ""
|
||||
finish_reason = None
|
||||
async for c in stream:
|
||||
if c.choices:
|
||||
if c.choices[0].delta.content:
|
||||
text += c.choices[0].delta.content
|
||||
if c.choices[0].finish_reason:
|
||||
finish_reason = c.choices[0].finish_reason
|
||||
|
||||
assert text == "one two three"
|
||||
assert finish_reason == "stop"
|
||||
expected_key: Final = explicit_key or (API_KEY if credential_source == "environment" else None)
|
||||
assert captured["authorization"] == (f"Bearer {expected_key}" if expected_key else None)
|
||||
|
||||
|
||||
def test_completion_streaming_tool_call_reassembly():
|
||||
"""Tool calls split across chunks must be correctly passed through by the OpenAI-compatible stream processor."""
|
||||
chunks = [
|
||||
{
|
||||
"id": "chatcmpl-test",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 1,
|
||||
"model": "clinepass/deepseek-v4-flash",
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"delta": {
|
||||
"tool_calls": [
|
||||
{
|
||||
"index": 0,
|
||||
"id": "call_123",
|
||||
"type": "function",
|
||||
"function": {"name": "get_weather", "arguments": ""},
|
||||
}
|
||||
]
|
||||
},
|
||||
"finish_reason": None,
|
||||
}
|
||||
],
|
||||
},
|
||||
{
|
||||
"id": "chatcmpl-test",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 1,
|
||||
"model": "clinepass/deepseek-v4-flash",
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"delta": {"tool_calls": [{"index": 0, "function": {"arguments": '{"loc'}}]},
|
||||
"finish_reason": None,
|
||||
}
|
||||
],
|
||||
},
|
||||
{
|
||||
"id": "chatcmpl-test",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 1,
|
||||
"model": "clinepass/deepseek-v4-flash",
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"delta": {"tool_calls": [{"index": 0, "function": {"arguments": 'ation": "NYC"}'}}]},
|
||||
"finish_reason": None,
|
||||
}
|
||||
],
|
||||
},
|
||||
{
|
||||
"id": "chatcmpl-test",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 1,
|
||||
"model": "clinepass/deepseek-v4-flash",
|
||||
"choices": [{"index": 0, "delta": {}, "finish_reason": "tool_calls"}],
|
||||
},
|
||||
]
|
||||
body = "".join(f"data: {json.dumps(c)}\n\n" for c in chunks) + "data: [DONE]\n\n"
|
||||
|
||||
def fake_post(self, url, *args, **kwargs):
|
||||
return httpx.Response(
|
||||
200,
|
||||
content=body.encode(),
|
||||
headers={"content-type": "text/event-stream"},
|
||||
request=httpx.Request("POST", str(url)),
|
||||
)
|
||||
|
||||
with patch.object(HTTPHandler, "post", fake_post):
|
||||
stream = litellm.completion(
|
||||
model="clinepass/deepseek-v4-flash",
|
||||
messages=[{"role": "user", "content": "weather"}],
|
||||
stream=True,
|
||||
)
|
||||
|
||||
args_text = ""
|
||||
for c in stream:
|
||||
if c.choices and c.choices[0].delta.tool_calls:
|
||||
tc = c.choices[0].delta.tool_calls[0]
|
||||
if tc.function and tc.function.arguments:
|
||||
args_text += tc.function.arguments
|
||||
|
||||
assert args_text == '{"location": "NYC"}'
|
||||
|
||||
|
||||
def test_completion_preserves_usage():
|
||||
def fake_post(self, url, *args, **kwargs):
|
||||
return _response(ENVELOPED_COMPLETION)
|
||||
|
||||
with patch.object(HTTPHandler, "post", fake_post):
|
||||
response = litellm.completion(
|
||||
model="clinepass/deepseek-v4-flash",
|
||||
messages=[{"role": "user", "content": "ping"}],
|
||||
)
|
||||
|
||||
assert response.usage.prompt_tokens == 5
|
||||
assert response.usage.completion_tokens == 2
|
||||
assert response.usage.total_tokens == 7
|
||||
|
||||
|
||||
def test_completion_sends_authorization_header():
|
||||
captured_headers = {}
|
||||
|
||||
def fake_post(self, url, *args, **kwargs):
|
||||
captured_headers.update(kwargs.get("headers", {}))
|
||||
return _response(ENVELOPED_COMPLETION)
|
||||
|
||||
with patch.object(HTTPHandler, "post", fake_post):
|
||||
litellm.completion(
|
||||
model="clinepass/deepseek-v4-flash",
|
||||
messages=[{"role": "user", "content": "ping"}],
|
||||
)
|
||||
|
||||
assert captured_headers.get("Authorization") == f"Bearer {API_KEY}"
|
||||
|
||||
|
||||
def test_completion_explicit_api_key_precedence():
|
||||
captured_headers = {}
|
||||
|
||||
def fake_post(self, url, *args, **kwargs):
|
||||
captured_headers.update(kwargs.get("headers", {}))
|
||||
return _response(ENVELOPED_COMPLETION)
|
||||
|
||||
with patch.object(HTTPHandler, "post", fake_post):
|
||||
litellm.completion(
|
||||
model="clinepass/deepseek-v4-flash",
|
||||
messages=[{"role": "user", "content": "ping"}],
|
||||
api_key="sk-clinepass-explicit-key",
|
||||
)
|
||||
|
||||
assert captured_headers.get("Authorization") == "Bearer sk-clinepass-explicit-key"
|
||||
|
||||
|
||||
def test_upstream_429_maps_to_rate_limit_error():
|
||||
from litellm.exceptions import RateLimitError
|
||||
|
||||
def fake_post(self, url, *args, **kwargs):
|
||||
raise httpx.HTTPStatusError(
|
||||
"Too Many Requests",
|
||||
request=httpx.Request("POST", str(url)),
|
||||
response=httpx.Response(
|
||||
429,
|
||||
json={"error": "Rate limit exceeded"},
|
||||
request=httpx.Request("POST", str(url)),
|
||||
),
|
||||
)
|
||||
|
||||
with patch.object(HTTPHandler, "post", fake_post), pytest.raises(RateLimitError) as excinfo:
|
||||
litellm.completion(
|
||||
model="clinepass/deepseek-v4-flash",
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
)
|
||||
|
||||
assert excinfo.value.status_code == 429
|
||||
|
||||
|
||||
def test_upstream_401_maps_to_authentication_error():
|
||||
"""ClinePass answers a bad key with HTTP 401; that must not be flattened
|
||||
into a generic APIConnectionError."""
|
||||
from litellm.exceptions import AuthenticationError
|
||||
|
||||
def fake_post(self, url, *args, **kwargs):
|
||||
raise httpx.HTTPStatusError(
|
||||
"Unauthorized",
|
||||
request=httpx.Request("POST", str(url)),
|
||||
response=httpx.Response(
|
||||
401,
|
||||
json={"error": "Unauthorized"},
|
||||
request=httpx.Request("POST", str(url)),
|
||||
),
|
||||
)
|
||||
|
||||
with patch.object(HTTPHandler, "post", fake_post), pytest.raises(AuthenticationError) as excinfo:
|
||||
litellm.completion(
|
||||
model="clinepass/deepseek-v4-flash",
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
max_tokens=100,
|
||||
)
|
||||
|
||||
assert excinfo.value.status_code == 401
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Truncation reporting
|
||||
#
|
||||
# ClinePass was once observed returning finish_reason "stop" on a completion cut
|
||||
# off by max_tokens. Re-probing the live API on 2026-08-22 could not reproduce
|
||||
# it, so the provider now reports the upstream finish reason unmodified to avoid
|
||||
# false positives on natural completions that land exactly on the cap.
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _truncated_envelope(completion_tokens: int, finish_reason: str = "stop") -> dict:
|
||||
payload = json.loads(json.dumps(ENVELOPED_COMPLETION))
|
||||
payload["data"]["choices"][0]["finish_reason"] = finish_reason
|
||||
payload["data"]["usage"]["completion_tokens"] = completion_tokens
|
||||
return payload
|
||||
|
||||
|
||||
def _complete(payload: dict, **kwargs):
|
||||
def fake_post(self, url, *args, **post_kwargs):
|
||||
return _response(payload)
|
||||
|
||||
with patch.object(HTTPHandler, "post", fake_post):
|
||||
return litellm.completion(
|
||||
model="clinepass/deepseek-v4-flash",
|
||||
messages=[{"role": "user", "content": "ping"}],
|
||||
**kwargs,
|
||||
)
|
||||
|
||||
|
||||
def test_completion_at_the_cap_preserves_upstream_stop():
|
||||
response = _complete(_truncated_envelope(4000), max_tokens=4000)
|
||||
assert response.choices[0].finish_reason == "stop"
|
||||
|
||||
|
||||
def test_completion_over_the_cap_preserves_upstream_stop():
|
||||
response = _complete(_truncated_envelope(4001), max_tokens=4000)
|
||||
assert response.choices[0].finish_reason == "stop"
|
||||
|
||||
|
||||
def test_completion_below_the_cap_keeps_stop():
|
||||
response = _complete(_truncated_envelope(3999), max_tokens=4000)
|
||||
assert response.choices[0].finish_reason == "stop"
|
||||
|
||||
|
||||
def test_upstream_length_is_left_alone():
|
||||
response = _complete(_truncated_envelope(4000, finish_reason="length"), max_tokens=4000)
|
||||
assert response.choices[0].finish_reason == "length"
|
||||
|
||||
|
||||
def test_no_max_tokens_means_no_rewrite():
|
||||
response = _complete(_truncated_envelope(4000))
|
||||
assert response.choices[0].finish_reason == "stop"
|
||||
|
||||
|
||||
def test_max_completion_tokens_param_preserves_upstream_stop():
|
||||
"""``max_completion_tokens`` is mapped to ``max_tokens`` before ``request_data`` is built."""
|
||||
response = _complete(_truncated_envelope(4000), max_completion_tokens=4000)
|
||||
assert response.choices[0].finish_reason == "stop"
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Model catalog
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_get_models_returns_empty_without_calling_the_api():
|
||||
"""ClinePass has no /models endpoint (404), and the inherited OpenAI
|
||||
implementation would ask for it at the wrong path. It must not make the
|
||||
request at all."""
|
||||
|
||||
def explode(*args, **kwargs): # pragma: no cover - must never run
|
||||
raise AssertionError("get_models() must not perform an HTTP request")
|
||||
|
||||
with patch.object(litellm.module_level_client, "get", explode):
|
||||
assert ClinePassConfig().get_models(api_key=API_KEY) == []
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# httpx internals
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_unwrap_envelope_survives_a_response_with_no_request_attached():
|
||||
"""`httpx.Response.request` RAISES RuntimeError rather than returning None
|
||||
when no request is attached, so the unwrap must ask for it defensively."""
|
||||
raw = httpx.Response(200, json=ENVELOPED_COMPLETION)
|
||||
unwrapped = _unwrap_response_envelope(raw)
|
||||
assert unwrapped.json() == ENVELOPED_COMPLETION["data"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_acompletion_uses_sync_transform_request_via_behaviour():
|
||||
"""BaseLLMHTTPHandler builds the body with the sync transform_request on both
|
||||
paths, so an async override would be dead code -- the shape of bug this
|
||||
provider already shipped once. We assert this by verifying `acompletion`
|
||||
invokes the sync transform (which we mock here to prove it runs)."""
|
||||
captured = {}
|
||||
|
||||
# We patch the sync transform_request to prove it is the one called
|
||||
# during the async flow.
|
||||
original_transform = ClinePassConfig().transform_request
|
||||
|
||||
def mock_transform_request(*args, **kwargs):
|
||||
captured["called"] = True
|
||||
return original_transform(*args, **kwargs)
|
||||
|
||||
async def fake_post(self, url, *args, **kwargs):
|
||||
return _response(ENVELOPED_COMPLETION)
|
||||
|
||||
with (
|
||||
patch.object(ClinePassConfig, "transform_request", side_effect=mock_transform_request),
|
||||
patch.object(AsyncHTTPHandler, "post", fake_post),
|
||||
):
|
||||
await litellm.acompletion(
|
||||
model="clinepass/deepseek-v4-flash",
|
||||
messages=[{"role": "user", "content": "ping"}],
|
||||
)
|
||||
|
||||
assert captured.get("called") is True
|
||||
259
tests/unit/llms/clinepass/test_clinepass_endpoint_guard.py
Normal file
259
tests/unit/llms/clinepass/test_clinepass_endpoint_guard.py
Normal file
|
|
@ -0,0 +1,259 @@
|
|||
"""ClinePass must not be reachable on endpoints it does not implement.
|
||||
|
||||
ClinePass implements chat completions only. Before this guard existed, listing
|
||||
it in `litellm.openai_compatible_providers` made the speech, transcription and
|
||||
image-generation branches in litellm match it. Those branches send the request
|
||||
to the provider's own `api_base` but read the credential from `OPENAI_API_KEY`,
|
||||
so a `litellm.speech(model="clinepass/...")` call POSTed the caller's OpenAI key
|
||||
to the Cline host -- for an endpoint that does not exist there.
|
||||
|
||||
Asserting "not in the list" is a structural check and lives with the other
|
||||
registry tests. These tests assert the behaviour instead: that no HTTP request
|
||||
leaves the process at all, and that the OpenAI credential is never transmitted.
|
||||
"""
|
||||
|
||||
import contextlib
|
||||
from typing import Final
|
||||
|
||||
import httpx
|
||||
import pytest
|
||||
from openai import AsyncOpenAI, OpenAI
|
||||
|
||||
import litellm
|
||||
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler
|
||||
|
||||
SENTINEL_OPENAI_KEY = "sk-sentinel-openai-key-must-never-be-transmitted"
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def no_request_allowed(monkeypatch):
|
||||
"""Fail loudly if anything attempts an outbound request, and record it.
|
||||
|
||||
Patched at the httpx transport layer rather than at litellm's handlers, so a
|
||||
future dispatch path that bypasses HTTPHandler is still caught.
|
||||
"""
|
||||
attempted: list[tuple[str, str]] = []
|
||||
|
||||
def record_and_block(self, request, *args, **kwargs):
|
||||
attempted.append((str(request.url), request.headers.get("authorization", "")))
|
||||
raise AssertionError(f"outbound request attempted to {request.url}")
|
||||
|
||||
def record_handler_and_block(self, url, *args, **kwargs):
|
||||
attempted.append((str(url), httpx.Headers(kwargs.get("headers") or {}).get("authorization", "")))
|
||||
raise AssertionError(f"outbound request attempted to {url}")
|
||||
|
||||
async def record_async_handler_and_block(self, url, *args, **kwargs):
|
||||
record_handler_and_block(self, url, *args, **kwargs)
|
||||
|
||||
monkeypatch.setattr(httpx.Client, "send", record_and_block, raising=True)
|
||||
monkeypatch.setattr(httpx.AsyncClient, "send", record_and_block, raising=True)
|
||||
monkeypatch.setattr(HTTPHandler, "post", record_handler_and_block, raising=True)
|
||||
monkeypatch.setattr(AsyncHTTPHandler, "post", record_async_handler_and_block, raising=True)
|
||||
|
||||
monkeypatch.setenv("OPENAI_API_KEY", SENTINEL_OPENAI_KEY)
|
||||
monkeypatch.setenv("CLINEPASS_API_KEY", "cp-test-key")
|
||||
return attempted
|
||||
|
||||
|
||||
def test_speech_makes_no_outbound_request(no_request_allowed):
|
||||
# Which exception litellm raises for an unsupported endpoint is its business
|
||||
# and may change; that nothing is transmitted is the contract under test. The
|
||||
# fixture's AssertionError means the network WAS reached, so it must escape
|
||||
# rather than be swallowed as "some exception happened".
|
||||
try:
|
||||
litellm.speech(model="clinepass/deepseek-v4-flash", input="hi", voice="alloy")
|
||||
except AssertionError:
|
||||
raise
|
||||
except Exception: # noqa: S110 - deliberate; see above
|
||||
pass
|
||||
|
||||
assert no_request_allowed == []
|
||||
|
||||
|
||||
@pytest.mark.parametrize("clinepass_key", [None, "cp-test-key"])
|
||||
@pytest.mark.parametrize("custom_llm_provider", [None, "clinepass"])
|
||||
@pytest.mark.parametrize("explicit_key", [None, "cp-explicit-key"])
|
||||
def test_moderation_rejects_clinepass_before_credential_fallback(
|
||||
monkeypatch, no_request_allowed, clinepass_key, custom_llm_provider, explicit_key
|
||||
):
|
||||
if clinepass_key is None:
|
||||
monkeypatch.delenv("CLINEPASS_API_KEY", raising=False)
|
||||
monkeypatch.setattr(litellm, "api_key", SENTINEL_OPENAI_KEY)
|
||||
model: Final = "deepseek-v4-flash" if custom_llm_provider else "clinepass/deepseek-v4-flash"
|
||||
|
||||
with pytest.raises(litellm.BadRequestError, match="ClinePass does not support moderation endpoints"):
|
||||
litellm.moderation(model=model, input="hi", custom_llm_provider=custom_llm_provider, api_key=explicit_key)
|
||||
|
||||
assert no_request_allowed == []
|
||||
|
||||
|
||||
@pytest.mark.parametrize("clinepass_key", [None, "cp-test-key"])
|
||||
@pytest.mark.parametrize("custom_llm_provider", [None, "clinepass"])
|
||||
@pytest.mark.parametrize("explicit_key", [None, "cp-explicit-key"])
|
||||
@pytest.mark.asyncio
|
||||
async def test_async_moderation_rejects_clinepass_before_credential_fallback(
|
||||
monkeypatch, no_request_allowed, clinepass_key, custom_llm_provider, explicit_key
|
||||
):
|
||||
if clinepass_key is None:
|
||||
monkeypatch.delenv("CLINEPASS_API_KEY", raising=False)
|
||||
monkeypatch.setattr(litellm, "api_key", SENTINEL_OPENAI_KEY)
|
||||
model: Final = "deepseek-v4-flash" if custom_llm_provider else "clinepass/deepseek-v4-flash"
|
||||
|
||||
with pytest.raises(litellm.BadRequestError, match="ClinePass does not support moderation endpoints"):
|
||||
await litellm.amoderation(
|
||||
model=model, input="hi", custom_llm_provider=custom_llm_provider, api_key=explicit_key
|
||||
)
|
||||
|
||||
assert no_request_allowed == []
|
||||
|
||||
|
||||
@pytest.mark.parametrize("async_mode", [False, True])
|
||||
@pytest.mark.asyncio
|
||||
async def test_openai_moderation_remains_supported(async_mode):
|
||||
sent = []
|
||||
|
||||
def respond(request):
|
||||
sent.append((str(request.url), request.headers["authorization"]))
|
||||
return httpx.Response(
|
||||
200,
|
||||
json={
|
||||
"id": "modr-test",
|
||||
"model": "moderation-test",
|
||||
"results": [{"flagged": False, "categories": {"violence": False}, "category_scores": {"violence": 0}}],
|
||||
},
|
||||
)
|
||||
|
||||
transport: Final = httpx.MockTransport(respond)
|
||||
if async_mode:
|
||||
async with AsyncOpenAI(
|
||||
api_key="sk-moderation-test", http_client=httpx.AsyncClient(transport=transport)
|
||||
) as client:
|
||||
response = await litellm.amoderation(model="openai/moderation-test", input="hi", client=client)
|
||||
else:
|
||||
with OpenAI(api_key="sk-moderation-test", http_client=httpx.Client(transport=transport)) as client:
|
||||
response = litellm.moderation(model="moderation-test", input="hi", client=client)
|
||||
|
||||
assert sent == [("https://api.openai.com/v1/moderations", "Bearer sk-moderation-test")]
|
||||
assert response.id == "modr-test"
|
||||
assert response.results[0].flagged is False
|
||||
|
||||
|
||||
def test_transcription_makes_no_outbound_request(no_request_allowed, tmp_path):
|
||||
audio = tmp_path / "a.mp3"
|
||||
audio.write_bytes(b"\x00\x00")
|
||||
|
||||
with open(audio, "rb") as handle:
|
||||
try:
|
||||
litellm.transcription(model="clinepass/deepseek-v4-flash", file=handle)
|
||||
except AssertionError:
|
||||
raise
|
||||
except Exception: # noqa: S110 - deliberate; see test_speech_makes_no_outbound_request
|
||||
pass
|
||||
|
||||
assert no_request_allowed == []
|
||||
|
||||
|
||||
def test_image_generation_makes_no_outbound_request(no_request_allowed):
|
||||
"""Image generation must not reach the Cline host.
|
||||
|
||||
Note it does not raise either: litellm returns an empty `ImageResponse` for
|
||||
any provider with no image support. That is pre-existing upstream behaviour,
|
||||
not a ClinePass quirk -- `mistral`, which has the same shape ClinePass now
|
||||
has (own module, absent from `openai_compatible_providers`, explicitly
|
||||
registered for exception mapping), returns the same empty response with zero
|
||||
outbound requests. So this test asserts the property that is ours to keep:
|
||||
nothing is transmitted.
|
||||
"""
|
||||
litellm.image_generation(model="clinepass/deepseek-v4-flash", prompt="a cat")
|
||||
|
||||
assert no_request_allowed == []
|
||||
|
||||
|
||||
def test_openai_credential_is_never_transmitted(no_request_allowed):
|
||||
"""The point of the P1: whatever happens, the OpenAI key must not go out."""
|
||||
for call in (
|
||||
lambda: litellm.speech(model="clinepass/deepseek-v4-flash", input="hi", voice="alloy"),
|
||||
lambda: litellm.transcription(model="clinepass/deepseek-v4-flash", file=None),
|
||||
lambda: litellm.image_generation(model="clinepass/deepseek-v4-flash", prompt="a cat"),
|
||||
):
|
||||
# Whether each endpoint raises or returns an empty response is upstream's
|
||||
# business; that no credential leaves the process is ours.
|
||||
with contextlib.suppress(Exception):
|
||||
call()
|
||||
|
||||
leaked = [url for url, auth in no_request_allowed if SENTINEL_OPENAI_KEY in auth]
|
||||
assert leaked == [], f"OPENAI_API_KEY was transmitted to {leaked}"
|
||||
|
||||
|
||||
def test_unknown_kwargs_are_flattened_not_wrapped_in_extra_body():
|
||||
"""Unknown kwargs stay flat in the optional params.
|
||||
|
||||
For a provider listed in `openai_compatible_providers`, `get_optional_params`
|
||||
nests them under `extra_body`. `BaseLLMHTTPHandler` later merges that back
|
||||
into the request body, so the wire body does not distinguish the two cases;
|
||||
this params-level check is what fails if ClinePass drifts back into the list.
|
||||
"""
|
||||
params = litellm.utils.get_optional_params(
|
||||
model="cline-pass/deepseek-v4-flash",
|
||||
custom_llm_provider="clinepass",
|
||||
temperature=0.5,
|
||||
some_vendor_knob=7,
|
||||
)
|
||||
|
||||
assert "extra_body" not in params
|
||||
assert params["some_vendor_knob"] == 7
|
||||
|
||||
|
||||
def test_chat_does_not_fall_back_to_the_global_litellm_api_key(monkeypatch):
|
||||
"""`litellm.api_key` is the caller's general-purpose (usually OpenAI) key.
|
||||
|
||||
With no ClinePass credential configured, chat must not borrow it: doing so
|
||||
sends that key to the Cline host as a Bearer token.
|
||||
"""
|
||||
sent: list[tuple[str, str]] = []
|
||||
|
||||
def record_and_stop(self, *args, **kwargs):
|
||||
sent.append((str(kwargs.get("url")), str((kwargs.get("headers") or {}).get("Authorization", ""))))
|
||||
raise RuntimeError("stop before any network I/O")
|
||||
|
||||
monkeypatch.setattr(HTTPHandler, "post", record_and_stop, raising=True)
|
||||
monkeypatch.delenv("CLINEPASS_API_KEY", raising=False)
|
||||
monkeypatch.setattr(litellm, "api_key", SENTINEL_OPENAI_KEY)
|
||||
|
||||
with contextlib.suppress(Exception):
|
||||
litellm.completion(
|
||||
model="clinepass/deepseek-v4-flash",
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
num_retries=0,
|
||||
)
|
||||
|
||||
assert sent, "the chat request never reached the transport, so nothing was checked"
|
||||
assert all(SENTINEL_OPENAI_KEY not in authorization for _, authorization in sent)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("clinepass_key", [None, "cp-test-key"])
|
||||
@pytest.mark.parametrize("explicit_key", [None, "cp-explicit-key"])
|
||||
@pytest.mark.parametrize("endpoint", ["client_secret", "transcription_session", "calls"])
|
||||
@pytest.mark.asyncio
|
||||
async def test_realtime_rejects_clinepass_before_credential_fallback(
|
||||
monkeypatch, no_request_allowed, clinepass_key, explicit_key, endpoint
|
||||
):
|
||||
if clinepass_key is None:
|
||||
monkeypatch.delenv("CLINEPASS_API_KEY", raising=False)
|
||||
else:
|
||||
monkeypatch.setenv("CLINEPASS_API_KEY", clinepass_key)
|
||||
monkeypatch.setattr(litellm, "api_key", SENTINEL_OPENAI_KEY)
|
||||
monkeypatch.setattr(litellm, "openai_key", SENTINEL_OPENAI_KEY)
|
||||
kwargs: Final = {"model": "clinepass/deepseek-v4-flash", "api_key": explicit_key}
|
||||
request: Final = (
|
||||
litellm.acreate_realtime_client_secret(**kwargs)
|
||||
if endpoint == "client_secret"
|
||||
else litellm.acreate_realtime_transcription_session(**kwargs)
|
||||
if endpoint == "transcription_session"
|
||||
else litellm.arealtime_calls(openai_ephemeral_key=SENTINEL_OPENAI_KEY, sdp_body=b"v=0\r\n", **kwargs)
|
||||
)
|
||||
|
||||
with pytest.raises(litellm.BadRequestError, match="ClinePass does not support realtime endpoints"):
|
||||
await request
|
||||
|
||||
assert no_request_allowed == []
|
||||
|
|
@ -176,6 +176,7 @@ describe("provider_info_helpers", () => {
|
|||
Providers.AUTO_ROUTER,
|
||||
Providers.BYTEZ,
|
||||
Providers.CLARIFAI,
|
||||
Providers.CLINEPASS,
|
||||
Providers.Cognition,
|
||||
Providers.COMPACTIFAI,
|
||||
Providers.DATAROBOT,
|
||||
|
|
|
|||
|
|
@ -89,6 +89,7 @@ export enum Providers {
|
|||
Cerebras = "Cerebras",
|
||||
CHATGPT = "ChatGPT Subscription",
|
||||
CLARIFAI = "Clarifai",
|
||||
CLINEPASS = "ClinePass",
|
||||
CLOUDFLARE = "Cloudflare",
|
||||
CODESTRAL = "Codestral",
|
||||
Cognition = "Cognition",
|
||||
|
|
@ -208,6 +209,7 @@ export const provider_map: Record<string, string> = {
|
|||
Cerebras: "cerebras",
|
||||
CHATGPT: "chatgpt",
|
||||
CLARIFAI: "clarifai",
|
||||
CLINEPASS: "clinepass",
|
||||
CLOUDFLARE: "cloudflare",
|
||||
CODESTRAL: "codestral",
|
||||
Cognition: "cognition",
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue