From 2553cbd337da7a539ebbe186393de4499b5c840b Mon Sep 17 00:00:00 2001 From: Daniel JB Clark Date: Fri, 21 Aug 2026 22:36:24 -0400 Subject: [PATCH 01/20] feat(clinepass): add ClinePass provider ClinePass (the Cline API) is OpenAI-compatible apart from two quirks: 1. Non-streaming completions are nested under a `data` envelope -- `{"data": {"choices": [...]}, "success": true}` -- rather than returning `choices` at the top level. Against the openai SDK this surfaces as `r.choices` being None, not as an error. Streaming responses are *not* enveloped, so SSE needs no special handling. 2. A bare model id is rejected with HTTP 400 "invalid model format. Expected format: modelType/model", but LiteLLM strips its own `clinepass/` routing prefix before the request is built, so it has to be restored. Both are handled in ClinePassConfig, which inherits OpenAIGPTConfig and overrides only transform_request/async_transform_request (prefix) and transform_response (unwrap). Unwrapping rebuilds the httpx.Response around the inner object, so the inherited OpenAI response transform -- including its `reasoning` -> `reasoning_content` mapping, which Cline populates -- is reused rather than duplicated. A response transform is the reason this cannot be a declarative entry in litellm/llms/openai_like/providers.json: JSON-configured providers are dispatched to `_complete_custom_openai`, which builds the request via provider_config but parses the response with convert_to_model_response_object and never calls provider_config.transform_response. Under that dispatch the unwrap is unreachable, so ClinePass gets an explicit branch in main.py routing it to base_llm_http_handler.completion. `clinepass` is still listed in openai_compatible_providers, which is what routes an upstream 401 through _map_openai_exception to AuthenticationError; the explicit dispatch branch precedes that list's catch-all, so membership does not send it back to the SDK path. Tests drive litellm.completion()/acompletion() against a mocked transport rather than calling the transforms directly, so dispatch itself is covered -- a unit test that calls transform_response() proves the function is correct but not that anything invokes it. Co-Authored-By: Claude Opus 5 (cherry picked from commit 86b3486e7cc8834ec9067e45f966b18b7466ec3b) --- litellm/__init__.py | 3 + litellm/_lazy_imports_registry.py | 2 + litellm/constants.py | 2 + .../get_llm_provider_logic.py | 5 + litellm/llms/clinepass/chat/transformation.py | 209 +++++++++++++ litellm/llms/clinepass/common_utils.py | 7 + litellm/main.py | 53 ++++ litellm/types/utils.py | 1 + litellm/utils.py | 1 + .../chat/test_clinepass_transformation.py | 295 ++++++++++++++++++ 10 files changed, 578 insertions(+) create mode 100644 litellm/llms/clinepass/chat/transformation.py create mode 100644 litellm/llms/clinepass/common_utils.py create mode 100644 tests/test_litellm/llms/clinepass/chat/test_clinepass_transformation.py diff --git a/litellm/__init__.py b/litellm/__init__.py index b6428b51bfb..40f90c4743a 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -2051,6 +2051,9 @@ if TYPE_CHECKING: ) from .llms.heroku.chat.transformation import HerokuChatConfig as HerokuChatConfig from .llms.cometapi.chat.transformation import CometAPIConfig as CometAPIConfig + from .llms.clinepass.chat.transformation import ( + ClinePassConfig as ClinePassConfig, + ) from .llms.azure.chat.gpt_transformation import ( AzureOpenAIConfig as AzureOpenAIConfig, ) diff --git a/litellm/_lazy_imports_registry.py b/litellm/_lazy_imports_registry.py index fcd2eed5387..aa2e8467946 100644 --- a/litellm/_lazy_imports_registry.py +++ b/litellm/_lazy_imports_registry.py @@ -284,6 +284,7 @@ LLM_CONFIG_NAMES: Final = ( "AzureOpenAIAssistantsAPIConfig", "HerokuChatConfig", "CometAPIConfig", + "ClinePassConfig", "AzureOpenAIConfig", "AzureOpenAIGPT5Config", "AzureOpenAITextConfig", @@ -1117,6 +1118,7 @@ _LLM_CONFIGS_IMPORT_MAP: Final = { ), "HerokuChatConfig": (".llms.heroku.chat.transformation", "HerokuChatConfig"), "CometAPIConfig": (".llms.cometapi.chat.transformation", "CometAPIConfig"), + "ClinePassConfig": (".llms.clinepass.chat.transformation", "ClinePassConfig"), "AzureOpenAIConfig": (".llms.azure.chat.gpt_transformation", "AzureOpenAIConfig"), "AzureOpenAIGPT5Config": ( ".llms.azure.chat.gpt_5_transformation", diff --git a/litellm/constants.py b/litellm/constants.py index 49514fc4d0e..6e5de432c51 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -782,6 +782,7 @@ LITELLM_CHAT_PROVIDERS: Final = [ "lemonade", "docker_model_runner", "amazon_nova", + "clinepass", ] # Resolving these providers runs an OAuth device flow (their provider info IS the login), so any @@ -1041,6 +1042,7 @@ openai_compatible_providers: Final[list] = [ "scx-ai", "prism", "sail", + "clinepass", # ClinePass (Cline API) - has its own module; listed here for exception mapping ] OPENAI_AUDIO_TRANSCRIPTION_PROVIDERS: Final = frozenset({"openai"} | frozenset(openai_compatible_providers)) diff --git a/litellm/litellm_core_utils/get_llm_provider_logic.py b/litellm/litellm_core_utils/get_llm_provider_logic.py index d4642ae2aad..18aeb55d5ec 100644 --- a/litellm/litellm_core_utils/get_llm_provider_logic.py +++ b/litellm/litellm_core_utils/get_llm_provider_logic.py @@ -611,6 +611,11 @@ def _get_openai_compatible_provider_info( api_base, dynamic_api_key, ) = litellm.PerplexityChatConfig()._get_openai_compatible_provider_info(api_base, api_key) + elif custom_llm_provider == "clinepass": + ( + api_base, + dynamic_api_key, + ) = litellm.ClinePassConfig()._get_openai_compatible_provider_info(api_base, api_key) elif custom_llm_provider == "aiohttp_openai": return model, "aiohttp_openai", api_key, api_base elif custom_llm_provider == "anyscale": diff --git a/litellm/llms/clinepass/chat/transformation.py b/litellm/llms/clinepass/chat/transformation.py new file mode 100644 index 00000000000..7b077467ebd --- /dev/null +++ b/litellm/llms/clinepass/chat/transformation.py @@ -0,0 +1,209 @@ +""" +Support for ClinePass (the Cline API) `/v1/chat/completions` endpoint. + +ClinePass is OpenAI-compatible apart from two quirks, both handled here: + +1. Non-streaming completions are nested under a ``data`` envelope -- + ``{"data": {"choices": [...]}, "success": true}`` -- rather than returning + ``choices`` at the top level. Streaming responses are *not* wrapped, so the + inherited SSE handling needs no change. +2. A bare model id is rejected with HTTP 400 ``invalid model format. Expected + format: modelType/model``, but LiteLLM strips its own ``clinepass/`` routing + prefix before the request is built, so it has to be restored. + +Documentation: https://docs.cline.bot/ +""" + +import json +from typing import Any, List, Tuple, Union + +import httpx + +from litellm.llms.base_llm.chat.transformation import BaseLLMException +from litellm.secret_managers.main import get_secret_str +from litellm.types.llms.openai import AllMessageValues +from litellm.types.utils import ModelResponse + +from ...openai.chat.gpt_transformation import OpenAIGPTConfig +from ..common_utils import ClinePassException + +CLINEPASS_API_BASE = "https://api.cline.bot/api/v1" + +# ClinePass nests the completion under this key on non-streaming responses. +CLINEPASS_RESPONSE_ENVELOPE_KEY = "data" + +# The qualifier ClinePass requires on outbound model ids. +CLINEPASS_MODEL_PREFIX = "clinepass/" + +# Headers that describe the original byte stream and would be wrong once the +# body is rewritten by _unwrap_response_envelope(). +_BODY_SPECIFIC_HEADERS = ("content-length", "content-encoding") + + +def _unwrap_response_envelope(raw_response: httpx.Response) -> httpx.Response: + """Strip ClinePass's ``data`` wrapper off a JSON completion body. + + The OpenAI transforms read ``raw_response.json()`` directly, so rebuild the + response around the inner object rather than duplicating their bodies here. + + Returns the original response untouched whenever the body does not look like + a wrapped completion, so an already-OpenAI-shaped body -- or an error nested + under the same key -- is not mistaken for one. + """ + try: + payload = raw_response.json() + except (ValueError, httpx.StreamError): + # Not a JSON body, or a streaming response that has not been read -- + # either way there is no envelope to strip. + return raw_response + + if not isinstance(payload, dict) or "choices" in payload: + return raw_response + + inner = payload.get(CLINEPASS_RESPONSE_ENVELOPE_KEY) + if not isinstance(inner, dict) or "choices" not in inner: + return raw_response + + headers = {k: v for k, v in raw_response.headers.items() if k.lower() not in _BODY_SPECIFIC_HEADERS} + + return httpx.Response( + status_code=raw_response.status_code, + headers=headers, + content=json.dumps(inner).encode("utf-8"), + request=getattr(raw_response, "_request", None), + ) + + +def _apply_model_prefix(data: dict) -> dict: + """Restore the ``modelType/model`` qualifier on the outbound model id. + + Only prefix ids that lost their qualifier, so a cross-provider id + (``clinepass/openrouter/foo`` -> ``openrouter/foo``) is forwarded unchanged. + """ + model = data.get("model") + if isinstance(model, str) and "/" not in model: + data["model"] = f"{CLINEPASS_MODEL_PREFIX}{model}" + return data + + +class ClinePassConfig(OpenAIGPTConfig): + """ + ClinePass configuration, inheriting the OpenAI chat transforms. + + Overrides only the request/response points where ClinePass diverges; see the + module docstring for the two quirks. + """ + + def _get_openai_compatible_provider_info( + self, api_base: str | None, api_key: str | None + ) -> Tuple[str | None, str | None]: + api_base = api_base or get_secret_str("CLINEPASS_API_BASE") or CLINEPASS_API_BASE + dynamic_api_key = api_key or get_secret_str("CLINEPASS_API_KEY") + return api_base, dynamic_api_key + + def get_complete_url( + self, + api_base: str | None, + api_key: str | None, + model: str, + optional_params: dict, + litellm_params: dict, + stream: bool | None = None, + ) -> str: + if not api_base: + api_base = CLINEPASS_API_BASE + + api_base = api_base.rstrip("/") + if api_base.endswith("/chat/completions"): + return api_base + + return f"{api_base}/chat/completions" + + def map_openai_params( + self, + non_default_params: dict, + optional_params: dict, + model: str, + drop_params: bool, + ) -> dict: + """ClinePass takes the legacy ``max_tokens`` spelling only.""" + mapped_params = super().map_openai_params( + non_default_params=non_default_params, + optional_params=optional_params, + model=model, + drop_params=drop_params, + ) + if "max_completion_tokens" in mapped_params: + mapped_params["max_tokens"] = mapped_params.pop("max_completion_tokens") + return mapped_params + + def transform_request( + self, + model: str, + messages: List[AllMessageValues], + optional_params: dict, + litellm_params: dict, + headers: dict, + ) -> dict: + data = super().transform_request( + model=model, + messages=messages, + optional_params=optional_params, + litellm_params=litellm_params, + headers=headers, + ) + return _apply_model_prefix(data) + + async def async_transform_request( + self, + model: str, + messages: List[AllMessageValues], + optional_params: dict, + litellm_params: dict, + headers: dict, + ) -> dict: + data = await super().async_transform_request( + model=model, + messages=messages, + optional_params=optional_params, + litellm_params=litellm_params, + headers=headers, + ) + return _apply_model_prefix(data) + + def transform_response( + self, + model: str, + raw_response: httpx.Response, + model_response: ModelResponse, + logging_obj: Any, + request_data: dict, + messages: List[AllMessageValues], + optional_params: dict, + litellm_params: dict, + encoding: Any, + api_key: str | None = None, + json_mode: bool | None = None, + ) -> ModelResponse: + return super().transform_response( + model=model, + raw_response=_unwrap_response_envelope(raw_response), + model_response=model_response, + logging_obj=logging_obj, + request_data=request_data, + messages=messages, + optional_params=optional_params, + litellm_params=litellm_params, + encoding=encoding, + api_key=api_key, + json_mode=json_mode, + ) + + def get_error_class( + self, error_message: str, status_code: int, headers: Union[dict, httpx.Headers] + ) -> BaseLLMException: + return ClinePassException( + message=error_message, + status_code=status_code, + headers=headers, + ) diff --git a/litellm/llms/clinepass/common_utils.py b/litellm/llms/clinepass/common_utils.py new file mode 100644 index 00000000000..27d6db446eb --- /dev/null +++ b/litellm/llms/clinepass/common_utils.py @@ -0,0 +1,7 @@ +from litellm.llms.base_llm.chat.transformation import BaseLLMException + + +class ClinePassException(BaseLLMException): + """ClinePass exception handling class""" + + pass diff --git a/litellm/main.py b/litellm/main.py index a818213b861..789f8ae6913 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -2448,6 +2448,57 @@ def _complete_aiohttp_openai( ) +def _complete_clinepass(ctx: _CompletionDispatchContext) -> _CompletionDispatchResult: + acompletion = ctx.acompletion + api_base = ctx.api_base + api_key = ctx.api_key + client = ctx.client + custom_llm_provider = ctx.custom_llm_provider + headers = ctx.headers + litellm_params = ctx.litellm_params + logging = ctx.logging + messages = ctx.messages + model = ctx.model + model_response = ctx.model_response + optional_params = ctx.optional_params + provider_config = ctx.provider_config + shared_session = ctx.shared_session + stream = ctx.stream + timeout = ctx.timeout + + api_key = api_key or get_secret_str("CLINEPASS_API_KEY") or litellm.api_key + + api_base = ( + api_base or litellm.api_base or get_secret_str("CLINEPASS_API_BASE") or "https://api.cline.bot/api/v1" + ) + + ## COMPLETION CALL + response = base_llm_http_handler.completion( + model=model, + messages=messages, + headers=headers, + model_response=model_response, + api_key=api_key, + api_base=api_base, + acompletion=acompletion, + logging_obj=logging, + optional_params=optional_params, + litellm_params=litellm_params, + shared_session=shared_session, + timeout=timeout, + client=client, + custom_llm_provider=custom_llm_provider, + encoding=_get_encoding(), + stream=stream, + provider_config=provider_config, + ) + + ## LOGGING + logging.post_call(input=messages, api_key=api_key, original_response=response) + + return response + + def _complete_cometapi(ctx: _CompletionDispatchContext) -> _CompletionDispatchResult: acompletion: Final = ctx.acompletion api_base = ctx.api_base @@ -5919,6 +5970,8 @@ def completion( response = _complete_aiohttp_openai(_dispatch_ctx) elif custom_llm_provider == "cometapi": response = _complete_cometapi(_dispatch_ctx) + elif custom_llm_provider == "clinepass": + response = _complete_clinepass(_dispatch_ctx) elif custom_llm_provider == "minimax": response = _complete_minimax(_dispatch_ctx) elif custom_llm_provider == "hosted_vllm": diff --git a/litellm/types/utils.py b/litellm/types/utils.py index a33eeaccaa3..754200cf5aa 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -4156,6 +4156,7 @@ class LlmProviders(str, Enum): APERTIS = "apertis" NANOGPT = "nano-gpt" POE = "poe" + CLINEPASS = "clinepass" CHUTES = "chutes" NEOSANTARA = "neosantara" PARASAIL = "parasail" diff --git a/litellm/utils.py b/litellm/utils.py index d72588e2b00..ab840f4016b 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -8537,6 +8537,7 @@ class ProviderConfigManager: LlmProviders.EDENAI: (litellm.EdenAIChatConfig, False), LlmProviders.FAL_AI: (litellm.FalAIChatConfig, False), LlmProviders.COMETAPI: (lambda: litellm.CometAPIConfig(), False), + LlmProviders.CLINEPASS: (lambda: litellm.ClinePassConfig(), False), LlmProviders.DATAROBOT: (lambda: litellm.DataRobotConfig(), False), LlmProviders.GEMINI: (lambda: litellm.GoogleAIStudioGeminiConfig(), False), LlmProviders.AI21: (lambda: litellm.AI21ChatConfig(), False), diff --git a/tests/test_litellm/llms/clinepass/chat/test_clinepass_transformation.py b/tests/test_litellm/llms/clinepass/chat/test_clinepass_transformation.py new file mode 100644 index 00000000000..e29968dcda0 --- /dev/null +++ b/tests/test_litellm/llms/clinepass/chat/test_clinepass_transformation.py @@ -0,0 +1,295 @@ +"""Tests for the ClinePass provider. + +The point of the end-to-end tests here is that they drive ``litellm.completion()`` +with a mocked transport rather than calling the transforms directly -- a unit test +that calls ``transform_response()`` itself proves the function is correct but not +that anything invokes it, which is exactly how the envelope unwrap was previously +shipped as dead code. +""" + +import json +from unittest.mock import patch + +import httpx +import pytest + +import litellm +from litellm.llms.clinepass.chat.transformation import ( + ClinePassConfig, + _apply_model_prefix, + _unwrap_response_envelope, +) +from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler +from litellm.types.utils import LlmProviders +from litellm.utils import ProviderConfigManager + +API_KEY = "sk-clinepass-test-not-real" + +ENVELOPED_COMPLETION = { + "success": True, + "data": { + "id": "chatcmpl-test", + "object": "chat.completion", + "created": 1, + "model": "clinepass/deepseek-v4-flash", + "choices": [ + { + "index": 0, + "finish_reason": "stop", + "message": { + "role": "assistant", + "content": "pong", + "reasoning": "the user asked for pong", + }, + } + ], + "usage": {"prompt_tokens": 5, "completion_tokens": 2, "total_tokens": 7}, + }, +} + + +def _response(payload: dict, url: str = "https://api.cline.bot/api/v1/chat/completions") -> httpx.Response: + return httpx.Response(200, json=payload, request=httpx.Request("POST", url)) + + +@pytest.fixture(autouse=True) +def _clinepass_env(monkeypatch): + monkeypatch.setenv("CLINEPASS_API_KEY", API_KEY) + monkeypatch.delenv("CLINEPASS_API_BASE", raising=False) + + +# -------------------------------------------------------------------------- +# Registration / routing +# -------------------------------------------------------------------------- + + +def test_get_llm_provider_resolves_clinepass(): + model, provider, api_key, api_base = litellm.get_llm_provider(model="clinepass/deepseek-v4-flash") + assert model == "deepseek-v4-flash" + assert provider == "clinepass" + assert api_key == API_KEY + assert api_base == "https://api.cline.bot/api/v1" + + +def test_provider_config_manager_returns_clinepass_config(): + config = ProviderConfigManager.get_provider_chat_config(model="deepseek-v4-flash", provider=LlmProviders.CLINEPASS) + assert isinstance(config, ClinePassConfig) + + +def test_clinepass_is_not_a_json_configured_provider(): + """ClinePass needs a response transform, which the JSON provider system's + OpenAI-SDK dispatch path never invokes. Guard against it drifting back.""" + from litellm.llms.openai_like.json_loader import JSONProviderRegistry + + assert not JSONProviderRegistry.exists("clinepass") + + +def test_clinepass_stays_in_openai_compatible_providers(): + """Membership drives `_map_openai_exception`, so dropping it silently + downgrades a 401 to APIConnectionError. The explicit dispatch branch in + main.py precedes the openai_compatible_providers catch-all, so being listed + here does NOT route ClinePass to the OpenAI SDK path.""" + assert "clinepass" in litellm.openai_compatible_providers + + +def test_api_base_env_override(monkeypatch): + monkeypatch.setenv("CLINEPASS_API_BASE", "https://proxy.internal/api/v1") + _, _, _, api_base = litellm.get_llm_provider(model="clinepass/deepseek-v4-flash") + assert api_base == "https://proxy.internal/api/v1" + + +@pytest.mark.parametrize( + "api_base,expected", + [ + (None, "https://api.cline.bot/api/v1/chat/completions"), + ("https://api.cline.bot/api/v1", "https://api.cline.bot/api/v1/chat/completions"), + ("https://api.cline.bot/api/v1/", "https://api.cline.bot/api/v1/chat/completions"), + ( + "https://api.cline.bot/api/v1/chat/completions", + "https://api.cline.bot/api/v1/chat/completions", + ), + ], +) +def test_get_complete_url(api_base, expected): + url = ClinePassConfig().get_complete_url( + api_base=api_base, api_key=API_KEY, model="deepseek-v4-flash", optional_params={}, litellm_params={} + ) + assert url == expected + + +# -------------------------------------------------------------------------- +# Model prefix +# -------------------------------------------------------------------------- + + +def test_model_prefix_restored_on_bare_id(): + assert _apply_model_prefix({"model": "deepseek-v4-flash"})["model"] == "clinepass/deepseek-v4-flash" + + +def test_model_prefix_left_alone_when_qualifier_present(): + """`clinepass/openrouter/foo` arrives here as `openrouter/foo` and must pass through.""" + assert _apply_model_prefix({"model": "openrouter/foo"})["model"] == "openrouter/foo" + + +def test_model_prefix_ignores_missing_model(): + assert _apply_model_prefix({}) == {} + + +# -------------------------------------------------------------------------- +# Response envelope +# -------------------------------------------------------------------------- + + +def test_unwrap_envelope_extracts_inner_completion(): + unwrapped = _unwrap_response_envelope(_response(ENVELOPED_COMPLETION)) + assert unwrapped.json() == ENVELOPED_COMPLETION["data"] + + +def test_unwrap_envelope_content_length_describes_the_new_body(): + """The original content-length describes the enveloped bytes and must not be + carried over; httpx recomputes a correct one for the rewritten body.""" + raw = _response(ENVELOPED_COMPLETION) + unwrapped = _unwrap_response_envelope(raw) + assert unwrapped.headers["content-length"] != raw.headers["content-length"] + assert int(unwrapped.headers["content-length"]) == len(unwrapped.content) + + +def test_unwrap_envelope_passes_through_openai_shaped_body(): + payload = ENVELOPED_COMPLETION["data"] + assert _unwrap_response_envelope(_response(payload)).json() == payload + + +def test_unwrap_envelope_passes_through_error_nested_under_same_key(): + """An error under `data` has no `choices` and must not be mistaken for a completion.""" + payload = {"success": False, "data": {"message": "bad model"}} + assert _unwrap_response_envelope(_response(payload)).json() == payload + + +def test_unwrap_envelope_passes_through_non_json_body(): + raw = httpx.Response( + 200, content=b"not json", request=httpx.Request("POST", "https://api.cline.bot/api/v1/chat/completions") + ) + assert _unwrap_response_envelope(raw) is raw + + +# -------------------------------------------------------------------------- +# Parameter mapping +# -------------------------------------------------------------------------- + + +def test_max_completion_tokens_mapped_to_max_tokens(): + mapped = ClinePassConfig().map_openai_params( + non_default_params={"max_completion_tokens": 4000}, + optional_params={}, + model="deepseek-v4-flash", + drop_params=False, + ) + assert mapped == {"max_tokens": 4000} + + +# -------------------------------------------------------------------------- +# End-to-end through litellm.completion() -- these are the load-bearing ones +# -------------------------------------------------------------------------- + + +def test_completion_unwraps_envelope_and_prefixes_model(): + captured = {} + + def fake_post(self, url, *args, **kwargs): + captured["url"] = str(url) + captured["body"] = json.loads(kwargs["data"]) + return _response(ENVELOPED_COMPLETION) + + with patch.object(HTTPHandler, "post", fake_post): + response = litellm.completion( + model="clinepass/deepseek-v4-flash", + messages=[{"role": "user", "content": "ping"}], + max_tokens=4000, + ) + + assert captured["url"] == "https://api.cline.bot/api/v1/chat/completions" + assert captured["body"]["model"] == "clinepass/deepseek-v4-flash" + assert response.choices[0].message.content == "pong" + assert response.choices[0].message.reasoning_content == "the user asked for pong" + + +@pytest.mark.asyncio +async def test_acompletion_unwraps_envelope_and_prefixes_model(): + captured = {} + + async def fake_post(self, url, *args, **kwargs): + captured["url"] = str(url) + captured["body"] = json.loads(kwargs["data"]) + return _response(ENVELOPED_COMPLETION) + + with patch.object(AsyncHTTPHandler, "post", fake_post): + response = await litellm.acompletion( + model="clinepass/deepseek-v4-flash", + messages=[{"role": "user", "content": "ping"}], + max_tokens=4000, + ) + + assert captured["url"] == "https://api.cline.bot/api/v1/chat/completions" + assert captured["body"]["model"] == "clinepass/deepseek-v4-flash" + assert response.choices[0].message.content == "pong" + + +def test_completion_streaming_is_not_unwrapped(): + """ClinePass does NOT wrap SSE chunks -- they are already OpenAI-shaped.""" + chunks = [ + { + "id": "chatcmpl-test", + "object": "chat.completion.chunk", + "created": 1, + "model": "clinepass/deepseek-v4-flash", + "choices": [{"index": 0, "delta": {"content": piece}, "finish_reason": None}], + } + for piece in ["one ", "two ", "three"] + ] + body = "".join(f"data: {json.dumps(c)}\n\n" for c in chunks) + "data: [DONE]\n\n" + + def fake_post(self, url, *args, **kwargs): + return httpx.Response( + 200, + content=body.encode(), + headers={"content-type": "text/event-stream"}, + request=httpx.Request("POST", str(url)), + ) + + with patch.object(HTTPHandler, "post", fake_post): + stream = litellm.completion( + model="clinepass/deepseek-v4-flash", + messages=[{"role": "user", "content": "count"}], + max_tokens=4000, + stream=True, + ) + text = "".join(c.choices[0].delta.content or "" for c in stream if c.choices) + + assert text == "one two three" + + +def test_upstream_401_maps_to_authentication_error(): + """ClinePass answers a bad key with HTTP 401; that must not be flattened + into a generic APIConnectionError.""" + from litellm.exceptions import AuthenticationError + + def fake_post(self, url, *args, **kwargs): + raise httpx.HTTPStatusError( + "Unauthorized", + request=httpx.Request("POST", str(url)), + response=httpx.Response( + 401, + json={"error": "Unauthorized"}, + request=httpx.Request("POST", str(url)), + ), + ) + + with patch.object(HTTPHandler, "post", fake_post): + with pytest.raises(AuthenticationError) as excinfo: + litellm.completion( + model="clinepass/deepseek-v4-flash", + messages=[{"role": "user", "content": "hi"}], + max_tokens=100, + ) + + assert excinfo.value.status_code == 401 From e688210c3f2e15c6bf35be0c277542ef8c7df61b Mon Sep 17 00:00:00 2001 From: Daniel JB Clark Date: Sat, 22 Aug 2026 12:00:24 -0400 Subject: [PATCH 02/20] fix(clinepass): correct the model namespace, harden truncation detection Addresses findings from three independent code reviews (Claude Opus 5, GPT-5.6-sol xhigh via codex, Grok 4.6 xhigh via cursor) ahead of proposing this branch upstream. Correctness: * The restored model qualifier was `clinepass/`, a namespace that does not exist. The catalog namespace is `cline-pass/` (hyphenated); `clinepass/` is LiteLLM's own routing prefix, which is stripped before the request is built. This failed silently because the API validates only the *shape* of a model id -- `totallybogus/deepseek-v4-flash` also returns HTTP 200. It is not inert, though: verified against the live API, `cline-pass/deepseek-v4-flash` resolves to `deepseek/deepseek-v4-flash` while any unrecognised namespace falls back to the date-pinned `deepseek/deepseek-v4-flash-0731`. * `_correct_truncated_finish_reason` compared an aggregate `usage.completion_tokens` against a per-choice `max_tokens`. With `n > 1` that relabels naturally-finished choices as truncated: two 60-token choices under a cap of 100 aggregate to 120 and both became `length`. Restricted to single-choice responses, where the inference is sound. * Zero, negative and `bool` caps are now rejected. `bool` subclasses `int`, so `max_tokens=True` was read as a cap of 1 and would relabel everything. Float usage/caps are now accepted rather than silently skipped. * `_unwrap_response_envelope` reached for the private `_request` attribute because `httpx.Response.request` raises `RuntimeError` instead of returning `None`. Ask for the public attribute defensively instead. * `get_models()` is overridden to return an empty catalog. ClinePass has no `/models` endpoint (404), and the inherited OpenAI implementation asked for it at the wrong path. * Dropped the `async_transform_request` override. `BaseLLMHTTPHandler` builds the body with the synchronous `transform_request` on both the sync and async paths, so it was dead code -- the same shape of bug this provider shipped once already. Pinned with a test. Packaging / CI: * Added the `clinepass` entry to `provider_endpoints_support.json` and its backup. Without it `check_provider_folders_documented.py` fails, which is a required Code Quality job -- verified failing before, passing after. * `transformation.py` did not satisfy `ruff format` under the repo's pinned ruff 0.15.3, which the changed-file CI gate runs. Reformatted. * This commit replaces the previous branch tip, which had accidentally swept in ~435 regenerated Next.js artifacts under `litellm/proxy/_experimental/`. The branch is now +883/-0 across 12 files. Honesty note on the truncation fix: re-probing the live API (streaming and non-streaming, caps of 2000 and 4000, both namespaces) could NOT reproduce the `stop`-instead-of-`length` misreport that motivated it. The correction is kept as a conservative safety net and documented as such rather than as a workaround for a currently-observable defect. Streaming is deliberately not covered; see the docstring. Tests: 41 pass (was 32). openai_like + cometapi regressions: 126 passed, 8 skipped. (cherry picked from commit c2dda4fd8b7a4e1874642b25d73656292cb52468) --- litellm/llms/clinepass/chat/transformation.py | 128 +++++++++--- .../provider_endpoints_support_backup.json | 18 ++ provider_endpoints_support.json | 18 ++ .../chat/test_clinepass_transformation.py | 193 +++++++++++++++++- 4 files changed, 331 insertions(+), 26 deletions(-) diff --git a/litellm/llms/clinepass/chat/transformation.py b/litellm/llms/clinepass/chat/transformation.py index 7b077467ebd..e0d61697ba6 100644 --- a/litellm/llms/clinepass/chat/transformation.py +++ b/litellm/llms/clinepass/chat/transformation.py @@ -9,13 +9,13 @@ ClinePass is OpenAI-compatible apart from two quirks, both handled here: inherited SSE handling needs no change. 2. A bare model id is rejected with HTTP 400 ``invalid model format. Expected format: modelType/model``, but LiteLLM strips its own ``clinepass/`` routing - prefix before the request is built, so it has to be restored. + prefix before the request is built, so a qualifier has to be restored. Documentation: https://docs.cline.bot/ """ import json -from typing import Any, List, Tuple, Union +from typing import Any, List, Optional, Tuple, Union import httpx @@ -32,14 +32,90 @@ CLINEPASS_API_BASE = "https://api.cline.bot/api/v1" # ClinePass nests the completion under this key on non-streaming responses. CLINEPASS_RESPONSE_ENVELOPE_KEY = "data" -# The qualifier ClinePass requires on outbound model ids. -CLINEPASS_MODEL_PREFIX = "clinepass/" +# The qualifier ClinePass expects on outbound model ids. +# +# Note the hyphen: the catalog namespace is ``cline-pass/``, not ``clinepass/`` +# (the latter is LiteLLM's own routing prefix, which is stripped before the +# request is built). The API only validates the *shape* of a model id -- any +# ``/`` is accepted with HTTP 200 -- so an incorrect namespace +# fails silently rather than loudly. It is not inert, though: for at least one +# model an unrecognised namespace resolves to a different, date-pinned snapshot +# (``cline-pass/deepseek-v4-flash`` -> ``deepseek/deepseek-v4-flash``, while +# ``clinepass/deepseek-v4-flash`` -> ``deepseek/deepseek-v4-flash-0731``). +CLINEPASS_MODEL_PREFIX = "cline-pass/" # Headers that describe the original byte stream and would be wrong once the # body is rewritten by _unwrap_response_envelope(). _BODY_SPECIFIC_HEADERS = ("content-length", "content-encoding") +def _as_positive_number(value: Any) -> Optional[float]: + """Return ``value`` as a positive number, or ``None`` if it is not one. + + ``bool`` is rejected explicitly: it is a subclass of ``int``, and ``True`` + would otherwise read as a cap of 1. + """ + if isinstance(value, bool) or not isinstance(value, (int, float)): + return None + if value <= 0: + return None + return float(value) + + +def _correct_truncated_finish_reason(response: ModelResponse, request_data: dict) -> ModelResponse: + """Report a truncated ClinePass completion as ``length``, not ``stop``. + + This is a defensive correction for an upstream bug in which ClinePass + returned ``finish_reason: "stop"`` on a completion that had actually been + cut off by ``max_tokens``: a request capped at 4000 came back with + ``completion_tokens == 4000`` and still claimed a natural stop. Callers that + trust ``finish_reason`` -- the documented way to detect truncation -- cannot + then distinguish a complete answer from a guillotined one. + + Re-probing the live API later (2026-08-22, both streaming and non-streaming, + caps of 2000 and 4000, across both the ``cline-pass/`` and the fallback + namespace) did *not* reproduce the misreport: every capped response + correctly returned ``length``. The upstream bug appears to have been fixed, + or to be intermittent. This correction is therefore kept as a cheap safety + net rather than as a workaround for a currently-observable defect, and it is + deliberately conservative: + + - only an upstream ``stop`` is ever rewritten; ``length`` is already right, + - only when usage shows the cap was actually reached, + - and only for single-choice responses. ``usage.completion_tokens`` is an + aggregate across all choices while ``max_tokens`` is a per-choice limit, + so with ``n > 1`` the aggregate cannot identify *which* choice was + truncated -- two naturally-finished 60-token choices under a cap of 100 + would otherwise both be relabelled ``length``. + + Streaming is deliberately not covered: chunks are assembled by the inherited + SSE iterator, the terminal ``finish_reason`` arrives before the usage chunk + that would justify rewriting it, and callers that omit + ``stream_options.include_usage`` never receive usage at all. Since the + misreport no longer reproduces, buffering the stream to correct it is not + worth the latency and complexity. + """ + if len(response.choices) != 1: + return response + + max_tokens = _as_positive_number(request_data.get("max_tokens")) + if max_tokens is None: + max_tokens = _as_positive_number(request_data.get("max_completion_tokens")) + if max_tokens is None: + return response + + usage = getattr(response, "usage", None) + completion_tokens = _as_positive_number(getattr(usage, "completion_tokens", None)) + if completion_tokens is None or completion_tokens < max_tokens: + return response + + for choice in response.choices: + if getattr(choice, "finish_reason", None) == "stop": + choice.finish_reason = "length" + + return response + + def _unwrap_response_envelope(raw_response: httpx.Response) -> httpx.Response: """Strip ClinePass's ``data`` wrapper off a JSON completion body. @@ -66,11 +142,19 @@ def _unwrap_response_envelope(raw_response: httpx.Response) -> httpx.Response: headers = {k: v for k, v in raw_response.headers.items() if k.lower() not in _BODY_SPECIFIC_HEADERS} + # httpx.Response.request raises RuntimeError rather than returning None when + # no request is attached, so ask for it defensively instead of reaching for + # the private attribute behind it. + try: + original_request = raw_response.request + except RuntimeError: + original_request = None + return httpx.Response( status_code=raw_response.status_code, headers=headers, content=json.dumps(inner).encode("utf-8"), - request=getattr(raw_response, "_request", None), + request=original_request, ) @@ -119,6 +203,17 @@ class ClinePassConfig(OpenAIGPTConfig): return f"{api_base}/chat/completions" + def get_models(self, api_key: Optional[str] = None, api_base: Optional[str] = None) -> List[str]: + """ClinePass exposes no model catalog. + + ``GET https://api.cline.bot/api/v1/models`` returns HTTP 404, and the + inherited OpenAI implementation would additionally ask for it at the + wrong path -- it rewrites the base URL down to scheme+host and appends + ``/v1/models``. Return an empty catalog rather than making a request + that is known to fail. + """ + return [] + def map_openai_params( self, non_default_params: dict, @@ -145,6 +240,9 @@ class ClinePassConfig(OpenAIGPTConfig): litellm_params: dict, headers: dict, ) -> dict: + # BaseLLMHTTPHandler builds the body with this synchronous method on + # both the sync and the async path, so there is deliberately no + # async_transform_request() override -- it would never be called. data = super().transform_request( model=model, messages=messages, @@ -154,23 +252,6 @@ class ClinePassConfig(OpenAIGPTConfig): ) return _apply_model_prefix(data) - async def async_transform_request( - self, - model: str, - messages: List[AllMessageValues], - optional_params: dict, - litellm_params: dict, - headers: dict, - ) -> dict: - data = await super().async_transform_request( - model=model, - messages=messages, - optional_params=optional_params, - litellm_params=litellm_params, - headers=headers, - ) - return _apply_model_prefix(data) - def transform_response( self, model: str, @@ -185,7 +266,7 @@ class ClinePassConfig(OpenAIGPTConfig): api_key: str | None = None, json_mode: bool | None = None, ) -> ModelResponse: - return super().transform_response( + response = super().transform_response( model=model, raw_response=_unwrap_response_envelope(raw_response), model_response=model_response, @@ -198,6 +279,7 @@ class ClinePassConfig(OpenAIGPTConfig): api_key=api_key, json_mode=json_mode, ) + return _correct_truncated_finish_reason(response, request_data) def get_error_class( self, error_message: str, status_code: int, headers: Union[dict, httpx.Headers] diff --git a/litellm/provider_endpoints_support_backup.json b/litellm/provider_endpoints_support_backup.json index c9635587eeb..8f52571799a 100644 --- a/litellm/provider_endpoints_support_backup.json +++ b/litellm/provider_endpoints_support_backup.json @@ -492,6 +492,24 @@ "interactions": true } }, + "clinepass": { + "display_name": "ClinePass (`clinepass`)", + "url": "https://docs.litellm.ai/docs/providers/clinepass", + "endpoints": { + "chat_completions": true, + "messages": true, + "responses": true, + "embeddings": false, + "image_generations": false, + "audio_transcriptions": false, + "audio_speech": false, + "moderations": false, + "batches": false, + "rerank": false, + "a2a": true, + "interactions": true + } + }, "cloudflare": { "display_name": "Cloudflare AI Workers (`cloudflare`)", "url": "https://docs.litellm.ai/docs/providers/cloudflare_workers", diff --git a/provider_endpoints_support.json b/provider_endpoints_support.json index 7ffaacdb3aa..1082d2046eb 100644 --- a/provider_endpoints_support.json +++ b/provider_endpoints_support.json @@ -546,6 +546,24 @@ "interactions": true } }, + "clinepass": { + "display_name": "ClinePass (`clinepass`)", + "url": "https://docs.litellm.ai/docs/providers/clinepass", + "endpoints": { + "chat_completions": true, + "messages": true, + "responses": true, + "embeddings": false, + "image_generations": false, + "audio_transcriptions": false, + "audio_speech": false, + "moderations": false, + "batches": false, + "rerank": false, + "a2a": true, + "interactions": true + } + }, "cloudflare": { "display_name": "Cloudflare AI Workers (`cloudflare`)", "url": "https://docs.litellm.ai/docs/providers/cloudflare_workers", diff --git a/tests/test_litellm/llms/clinepass/chat/test_clinepass_transformation.py b/tests/test_litellm/llms/clinepass/chat/test_clinepass_transformation.py index e29968dcda0..588aa7ec42f 100644 --- a/tests/test_litellm/llms/clinepass/chat/test_clinepass_transformation.py +++ b/tests/test_litellm/llms/clinepass/chat/test_clinepass_transformation.py @@ -17,6 +17,7 @@ import litellm from litellm.llms.clinepass.chat.transformation import ( ClinePassConfig, _apply_model_prefix, + _correct_truncated_finish_reason, _unwrap_response_envelope, ) from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler @@ -123,7 +124,14 @@ def test_get_complete_url(api_base, expected): def test_model_prefix_restored_on_bare_id(): - assert _apply_model_prefix({"model": "deepseek-v4-flash"})["model"] == "clinepass/deepseek-v4-flash" + """The restored qualifier is the catalog namespace ``cline-pass/`` (hyphenated), + NOT LiteLLM's own ``clinepass/`` routing prefix. + + The API validates only the *shape* of a model id, so a wrong namespace still + returns HTTP 200 -- but it does not always resolve to the same underlying + model, which makes a wrong value silent rather than harmless. + """ + assert _apply_model_prefix({"model": "deepseek-v4-flash"})["model"] == "cline-pass/deepseek-v4-flash" def test_model_prefix_left_alone_when_qualifier_present(): @@ -208,7 +216,7 @@ def test_completion_unwraps_envelope_and_prefixes_model(): ) assert captured["url"] == "https://api.cline.bot/api/v1/chat/completions" - assert captured["body"]["model"] == "clinepass/deepseek-v4-flash" + assert captured["body"]["model"] == "cline-pass/deepseek-v4-flash" assert response.choices[0].message.content == "pong" assert response.choices[0].message.reasoning_content == "the user asked for pong" @@ -230,7 +238,7 @@ async def test_acompletion_unwraps_envelope_and_prefixes_model(): ) assert captured["url"] == "https://api.cline.bot/api/v1/chat/completions" - assert captured["body"]["model"] == "clinepass/deepseek-v4-flash" + assert captured["body"]["model"] == "cline-pass/deepseek-v4-flash" assert response.choices[0].message.content == "pong" @@ -293,3 +301,182 @@ def test_upstream_401_maps_to_authentication_error(): ) assert excinfo.value.status_code == 401 + + +# -------------------------------------------------------------------------- +# Truncation reporting +# +# ClinePass was once observed returning finish_reason "stop" on a completion cut +# off by max_tokens. Re-probing the live API on 2026-08-22 could not reproduce +# it (see _correct_truncated_finish_reason's docstring), so the correction is a +# conservative safety net: single-choice only, upstream "stop" only, and only +# when usage shows the cap was actually reached. +# -------------------------------------------------------------------------- + + +def _truncated_envelope(completion_tokens: int, finish_reason: str = "stop") -> dict: + payload = json.loads(json.dumps(ENVELOPED_COMPLETION)) + payload["data"]["choices"][0]["finish_reason"] = finish_reason + payload["data"]["usage"]["completion_tokens"] = completion_tokens + return payload + + +def _complete(payload: dict, **kwargs): + def fake_post(self, url, *args, **post_kwargs): + return _response(payload) + + with patch.object(HTTPHandler, "post", fake_post): + return litellm.completion( + model="clinepass/deepseek-v4-flash", + messages=[{"role": "user", "content": "ping"}], + **kwargs, + ) + + +def test_completion_at_the_cap_is_reported_as_length_not_stop(): + response = _complete(_truncated_envelope(4000), max_tokens=4000) + assert response.choices[0].finish_reason == "length" + + +def test_completion_over_the_cap_is_reported_as_length(): + response = _complete(_truncated_envelope(4001), max_tokens=4000) + assert response.choices[0].finish_reason == "length" + + +def test_completion_below_the_cap_keeps_stop(): + response = _complete(_truncated_envelope(3999), max_tokens=4000) + assert response.choices[0].finish_reason == "stop" + + +def test_upstream_length_is_left_alone(): + response = _complete(_truncated_envelope(4000, finish_reason="length"), max_tokens=4000) + assert response.choices[0].finish_reason == "length" + + +def test_no_max_tokens_means_no_rewrite(): + response = _complete(_truncated_envelope(4000)) + assert response.choices[0].finish_reason == "stop" + + +def test_max_completion_tokens_also_detects_truncation(): + response = _complete(_truncated_envelope(4000), max_completion_tokens=4000) + assert response.choices[0].finish_reason == "length" + + +@pytest.mark.parametrize("bad", [None, "4000", 0, -1, True, False]) +def test_unusable_cap_is_ignored(bad): + """Non-numeric, zero, negative and bool caps carry no truncation signal. + + ``bool`` matters because it subclasses ``int``: ``True`` would otherwise be + read as a cap of 1 and relabel every response as truncated. + """ + + class _Choice: + finish_reason = "stop" + + class _Usage: + completion_tokens = 9999 + + class _Response: + choices = [_Choice()] + usage = _Usage() + + result = _correct_truncated_finish_reason(_Response(), {"max_tokens": bad}) + assert result.choices[0].finish_reason == "stop" + + +def test_missing_usage_is_ignored(): + class _Choice: + finish_reason = "stop" + + class _Response: + choices = [_Choice()] + usage = None + + result = _correct_truncated_finish_reason(_Response(), {"max_tokens": 4000}) + assert result.choices[0].finish_reason == "stop" + + +def _stub_response(finish_reasons, completion_tokens): + """Minimal ModelResponse-shaped stub for the truncation helper.""" + + class _Choice: + def __init__(self, reason): + self.finish_reason = reason + + class _Usage: + pass + + usage = _Usage() + usage.completion_tokens = completion_tokens + + class _Response: + pass + + response = _Response() + response.choices = [_Choice(r) for r in finish_reasons] + response.usage = usage + return response + + +def test_multi_choice_response_is_never_rewritten(): + """`usage.completion_tokens` is an aggregate across choices while `max_tokens` + is per choice, so the aggregate cannot say WHICH choice was truncated. + + Two naturally-finished 60-token choices under a cap of 100 aggregate to 120, + which would otherwise relabel both as `length`. + """ + response = _stub_response(["stop", "stop"], completion_tokens=120) + result = _correct_truncated_finish_reason(response, {"max_tokens": 100}) + assert [c.finish_reason for c in result.choices] == ["stop", "stop"] + + +def test_single_choice_at_the_cap_is_still_rewritten(): + """The n>1 guard must not disable the correction for the normal n=1 case.""" + response = _stub_response(["stop"], completion_tokens=100) + result = _correct_truncated_finish_reason(response, {"max_tokens": 100}) + assert result.choices[0].finish_reason == "length" + + +def test_float_usage_and_cap_are_honoured(): + """A gateway that reports usage as JSON floats must still be understood.""" + response = _stub_response(["stop"], completion_tokens=4000.0) + result = _correct_truncated_finish_reason(response, {"max_tokens": 4000.0}) + assert result.choices[0].finish_reason == "length" + + +# -------------------------------------------------------------------------- +# Model catalog +# -------------------------------------------------------------------------- + + +def test_get_models_returns_empty_without_calling_the_api(): + """ClinePass has no /models endpoint (404), and the inherited OpenAI + implementation would ask for it at the wrong path. It must not make the + request at all.""" + + def explode(*args, **kwargs): # pragma: no cover - must never run + raise AssertionError("get_models() must not perform an HTTP request") + + with patch.object(litellm.module_level_client, "get", explode): + assert ClinePassConfig().get_models(api_key=API_KEY) == [] + + +# -------------------------------------------------------------------------- +# httpx internals +# -------------------------------------------------------------------------- + + +def test_unwrap_envelope_survives_a_response_with_no_request_attached(): + """`httpx.Response.request` RAISES RuntimeError rather than returning None + when no request is attached, so the unwrap must ask for it defensively.""" + raw = httpx.Response(200, json=ENVELOPED_COMPLETION) + unwrapped = _unwrap_response_envelope(raw) + assert unwrapped.json() == ENVELOPED_COMPLETION["data"] + + +def test_clinepass_config_has_no_async_transform_request_override(): + """BaseLLMHTTPHandler builds the body with the sync transform_request on both + paths, so an async override would be dead code -- the shape of bug this + provider already shipped once.""" + assert "async_transform_request" not in ClinePassConfig.__dict__ From 9a91da39d38630204d8c0b49498bcd9ec0f41352 Mon Sep 17 00:00:00 2001 From: Daniel JB Clark Date: Sat, 22 Aug 2026 12:49:02 -0400 Subject: [PATCH 03/20] feat(clinepass): add Admin UI credential fields entry Co-Authored-By: Claude Fable 5 (cherry picked from commit 890301fa4fc25b5df46c735c7f2ca2c024c8ac63) --- .../provider_create_fields.json | 28 +++++++++++++++++++ 1 file changed, 28 insertions(+) diff --git a/litellm/proxy/public_endpoints/provider_create_fields.json b/litellm/proxy/public_endpoints/provider_create_fields.json index 6e96d6ad0ec..5135cb948db 100644 --- a/litellm/proxy/public_endpoints/provider_create_fields.json +++ b/litellm/proxy/public_endpoints/provider_create_fields.json @@ -829,6 +829,34 @@ ], "default_model_placeholder": "gpt-3.5-turbo" }, + { + "provider": "CLINEPASS", + "provider_display_name": "ClinePass", + "litellm_provider": "clinepass", + "credential_fields": [ + { + "key": "api_base", + "label": "API Base", + "placeholder": null, + "tooltip": null, + "required": false, + "field_type": "text", + "options": null, + "default_value": null + }, + { + "key": "api_key", + "label": "API Key", + "placeholder": null, + "tooltip": null, + "required": false, + "field_type": "password", + "options": null, + "default_value": null + } + ], + "default_model_placeholder": "gpt-3.5-turbo" + }, { "provider": "CLOUDFLARE", "provider_display_name": "Cloudflare", From 7d9ef1546d9cfe4121dcb3f9d54f91d476fea601 Mon Sep 17 00:00:00 2001 From: Daniel JB Clark Date: Sat, 3 Oct 2026 06:26:13 -0400 Subject: [PATCH 04/20] test(clinepass): move the provider tests into the CI-selected tree The tests were under tests/test_litellm/llms/clinepass/chat/, but .circleci/scripts/unit_selection.sh:62 builds the llm-other-providers shard with `find tests/unit/llms -name 'test_*.py'`, and GitHub Actions consumes the same script. Nothing errored -- the 482 lines of tests simply never ran in CI, so the PR would have looked green on tests that never executed. Moved to tests/unit/llms/clinepass/chat/, matching the layout cometapi and deepseek already use for an OpenAI-compatible provider, including the __init__.py files those directories carry, and renamed to test_clinepass_chat_transformation.py to match that same convention. 41 passed at the new location, where tests/unit/conftest.py applies rather than tests/test_litellm/conftest.py. The CI selector now lists the file. --- tests/unit/llms/clinepass/__init__.py | 0 tests/unit/llms/clinepass/chat/__init__.py | 0 .../llms/clinepass/chat/test_clinepass_chat_transformation.py} | 0 3 files changed, 0 insertions(+), 0 deletions(-) create mode 100644 tests/unit/llms/clinepass/__init__.py create mode 100644 tests/unit/llms/clinepass/chat/__init__.py rename tests/{test_litellm/llms/clinepass/chat/test_clinepass_transformation.py => unit/llms/clinepass/chat/test_clinepass_chat_transformation.py} (100%) diff --git a/tests/unit/llms/clinepass/__init__.py b/tests/unit/llms/clinepass/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/tests/unit/llms/clinepass/chat/__init__.py b/tests/unit/llms/clinepass/chat/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/tests/test_litellm/llms/clinepass/chat/test_clinepass_transformation.py b/tests/unit/llms/clinepass/chat/test_clinepass_chat_transformation.py similarity index 100% rename from tests/test_litellm/llms/clinepass/chat/test_clinepass_transformation.py rename to tests/unit/llms/clinepass/chat/test_clinepass_chat_transformation.py From b9d159fcc9651594d794ac5ba12a35b10319b3db Mon Sep 17 00:00:00 2001 From: Daniel JB Clark Date: Sat, 3 Oct 2026 06:54:20 -0400 Subject: [PATCH 05/20] fix(clinepass): stop unsupported endpoints reaching the provider with the OpenAI key ClinePass implements chat completions only, but it was listed in `openai_compatible_providers` to reach `_map_openai_exception`. That list is not inert for routing: * the speech branch in `main.py` matched it, and sends the request to the provider's own `api_base` while reading the credential from `OPENAI_API_KEY` -- so `litellm.speech(model="clinepass/...")` POSTed the caller's OpenAI key to the Cline host for an endpoint that does not exist there; * `OPENAI_AUDIO_TRANSCRIPTION_PROVIDERS` is derived from the same list, opening the identical hole for transcription; * `litellm/images/main.py` consults it for image generation; * `_add_provider_specific_params` packs unknown kwargs into `extra_body`, an OpenAI *SDK* concept the SDK unwraps client-side. ClinePass dispatches through `BaseLLMHTTPHandler`, which serialises optional params straight into the JSON body -- so membership put a literal `"extra_body": {}` on the wire on every chat request and buried genuine vendor kwargs one level deep. Drop the membership and register ClinePass explicitly for `_map_openai_exception` alongside `mistral` and `runwayml`, which is the established shape for a provider with its own module. Exception mapping is unchanged and still covered by `test_upstream_401_maps_to_authentication_error`. Verified: speech, transcription and image generation now make zero outbound requests and transmit no credential; unknown chat kwargs are flattened instead of wrapped. Image generation returns an empty `ImageResponse` rather than raising, which is pre-existing upstream behaviour for a provider with no image support -- `mistral` behaves identically. Co-Authored-By: Claude Opus 5 --- litellm/constants.py | 1 - .../exception_mapping_utils.py | 1 + .../test_clinepass_chat_transformation.py | 36 ++++- .../test_clinepass_endpoint_guard.py | 123 ++++++++++++++++++ 4 files changed, 153 insertions(+), 8 deletions(-) create mode 100644 tests/unit/llms/clinepass/test_clinepass_endpoint_guard.py diff --git a/litellm/constants.py b/litellm/constants.py index 6e5de432c51..372585cd5cc 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -1042,7 +1042,6 @@ openai_compatible_providers: Final[list] = [ "scx-ai", "prism", "sail", - "clinepass", # ClinePass (Cline API) - has its own module; listed here for exception mapping ] OPENAI_AUDIO_TRANSCRIPTION_PROVIDERS: Final = frozenset({"openai"} | frozenset(openai_compatible_providers)) diff --git a/litellm/litellm_core_utils/exception_mapping_utils.py b/litellm/litellm_core_utils/exception_mapping_utils.py index 0fdfb301291..6af3a8912a1 100644 --- a/litellm/litellm_core_utils/exception_mapping_utils.py +++ b/litellm/litellm_core_utils/exception_mapping_utils.py @@ -2468,6 +2468,7 @@ def exception_type( or custom_llm_provider == "text-completion-openai" or custom_llm_provider == "custom_openai" or custom_llm_provider in litellm.openai_compatible_providers + or custom_llm_provider == "clinepass" or custom_llm_provider == "mistral" or custom_llm_provider == "runwayml" ): diff --git a/tests/unit/llms/clinepass/chat/test_clinepass_chat_transformation.py b/tests/unit/llms/clinepass/chat/test_clinepass_chat_transformation.py index 588aa7ec42f..7c2c6792612 100644 --- a/tests/unit/llms/clinepass/chat/test_clinepass_chat_transformation.py +++ b/tests/unit/llms/clinepass/chat/test_clinepass_chat_transformation.py @@ -85,12 +85,33 @@ def test_clinepass_is_not_a_json_configured_provider(): assert not JSONProviderRegistry.exists("clinepass") -def test_clinepass_stays_in_openai_compatible_providers(): - """Membership drives `_map_openai_exception`, so dropping it silently - downgrades a 401 to APIConnectionError. The explicit dispatch branch in - main.py precedes the openai_compatible_providers catch-all, so being listed - here does NOT route ClinePass to the OpenAI SDK path.""" - assert "clinepass" in litellm.openai_compatible_providers +def test_clinepass_is_not_in_openai_compatible_providers(): + """ClinePass must NOT be in `openai_compatible_providers`. + + An earlier revision listed it there to reach `_map_openai_exception`, on the + assumption that the explicit dispatch branch in main.py made membership + inert for routing. It is not inert. The list is also consulted by: + + * the speech branch in `main.py` -- which sends the request to the + provider's own `api_base` while taking the key from + `OPENAI_API_KEY`, leaking the user's OpenAI credential to a third-party + host for an endpoint ClinePass does not even implement; + * `OPENAI_AUDIO_TRANSCRIPTION_PROVIDERS`, derived from this list, which + opens the same hole for transcription; + * image generation in `litellm/images/main.py`; + * `_add_provider_specific_params` in `litellm/utils.py`, which packs unknown + kwargs into `extra_body`. That is an OpenAI *SDK* concept the SDK unwraps + client-side. ClinePass dispatches through `BaseLLMHTTPHandler`, which + serialises optional params straight into the JSON body -- so membership + put a literal `"extra_body": {}` on the wire on every chat request, and + buried genuine vendor kwargs one level deep instead of flattening them. + + Exception mapping is preserved by registering ClinePass explicitly alongside + `mistral` and `runwayml` in `exception_mapping_utils.py`; + `test_upstream_401_maps_to_authentication_error` is the guard for that. + """ + assert "clinepass" not in litellm.openai_compatible_providers + assert "clinepass" not in litellm.constants.OPENAI_AUDIO_TRANSCRIPTION_PROVIDERS def test_api_base_env_override(monkeypatch): @@ -358,7 +379,8 @@ def test_no_max_tokens_means_no_rewrite(): assert response.choices[0].finish_reason == "stop" -def test_max_completion_tokens_also_detects_truncation(): +def test_max_completion_tokens_param_detects_truncation_after_mapping(): + """``max_completion_tokens`` is mapped to ``max_tokens`` before ``request_data`` is built.""" response = _complete(_truncated_envelope(4000), max_completion_tokens=4000) assert response.choices[0].finish_reason == "length" diff --git a/tests/unit/llms/clinepass/test_clinepass_endpoint_guard.py b/tests/unit/llms/clinepass/test_clinepass_endpoint_guard.py new file mode 100644 index 00000000000..83d798c9b35 --- /dev/null +++ b/tests/unit/llms/clinepass/test_clinepass_endpoint_guard.py @@ -0,0 +1,123 @@ +"""ClinePass must not be reachable on endpoints it does not implement. + +ClinePass implements chat completions only. Before this guard existed, listing +it in `litellm.openai_compatible_providers` made the speech, transcription and +image-generation branches in litellm match it. Those branches send the request +to the provider's own `api_base` but read the credential from `OPENAI_API_KEY`, +so a `litellm.speech(model="clinepass/...")` call POSTed the caller's OpenAI key +to the Cline host -- for an endpoint that does not exist there. + +Asserting "not in the list" is a structural check and lives with the other +registry tests. These tests assert the behaviour instead: that no HTTP request +leaves the process at all, and that the OpenAI credential is never transmitted. +""" + +import contextlib + +import httpx +import pytest + +import litellm +from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler + +SENTINEL_OPENAI_KEY = "sk-sentinel-openai-key-must-never-be-transmitted" + + +@pytest.fixture +def no_request_allowed(monkeypatch): + """Fail loudly if anything attempts an outbound request, and record it. + + Patched at the httpx transport layer rather than at litellm's handlers, so a + future dispatch path that bypasses HTTPHandler is still caught. + """ + attempted: list[tuple[str, str]] = [] + + def record_and_block(self, request, *args, **kwargs): + attempted.append((str(request.url), request.headers.get("authorization", ""))) + raise AssertionError(f"outbound request attempted to {request.url}") + + monkeypatch.setattr(httpx.Client, "send", record_and_block, raising=True) + monkeypatch.setattr(httpx.AsyncClient, "send", record_and_block, raising=True) + for handler in (HTTPHandler, AsyncHTTPHandler): + monkeypatch.setattr(handler, "post", record_and_block, raising=True) + + monkeypatch.setenv("OPENAI_API_KEY", SENTINEL_OPENAI_KEY) + monkeypatch.setenv("CLINEPASS_API_KEY", "cp-test-key") + return attempted + + +def test_speech_makes_no_outbound_request(no_request_allowed): + with pytest.raises(Exception) as excinfo: + litellm.speech(model="clinepass/deepseek-v4-flash", input="hi", voice="alloy") + + assert not isinstance(excinfo.value, AssertionError), ( + "speech() reached the network; the unsupported-endpoint guard is gone" + ) + assert no_request_allowed == [] + + +def test_transcription_makes_no_outbound_request(no_request_allowed, tmp_path): + audio = tmp_path / "a.mp3" + audio.write_bytes(b"\x00\x00") + + with open(audio, "rb") as handle: + with pytest.raises(Exception) as excinfo: + litellm.transcription(model="clinepass/deepseek-v4-flash", file=handle) + + assert not isinstance(excinfo.value, AssertionError) + assert no_request_allowed == [] + + +def test_image_generation_makes_no_outbound_request(no_request_allowed): + """Image generation must not reach the Cline host. + + Note it does not raise either: litellm returns an empty `ImageResponse` for + any provider with no image support. That is pre-existing upstream behaviour, + not a ClinePass quirk -- `mistral`, which has the same shape ClinePass now + has (own module, absent from `openai_compatible_providers`, explicitly + registered for exception mapping), returns the same empty response with zero + outbound requests. So this test asserts the property that is ours to keep: + nothing is transmitted. + """ + litellm.image_generation(model="clinepass/deepseek-v4-flash", prompt="a cat") + + assert no_request_allowed == [] + + +def test_openai_credential_is_never_transmitted(no_request_allowed): + """The point of the P1: whatever happens, the OpenAI key must not go out.""" + for call in ( + lambda: litellm.speech( + model="clinepass/deepseek-v4-flash", input="hi", voice="alloy" + ), + lambda: litellm.transcription(model="clinepass/deepseek-v4-flash", file=None), + lambda: litellm.image_generation( + model="clinepass/deepseek-v4-flash", prompt="a cat" + ), + ): + # Whether each endpoint raises or returns an empty response is upstream's + # business; that no credential leaves the process is ours. + with contextlib.suppress(Exception): + call() + + leaked = [url for url, auth in no_request_allowed if SENTINEL_OPENAI_KEY in auth] + assert leaked == [], f"OPENAI_API_KEY was transmitted to {leaked}" + + +def test_unknown_kwargs_are_flattened_not_wrapped_in_extra_body(): + """`extra_body` is an OpenAI-SDK concept the SDK unwraps client-side. + + ClinePass dispatches through `BaseLLMHTTPHandler`, which serialises optional + params directly into the JSON body, so an `extra_body` key would be sent to + the API verbatim and a genuine vendor kwarg would arrive nested one level + too deep. + """ + params = litellm.utils.get_optional_params( + model="cline-pass/deepseek-v4-flash", + custom_llm_provider="clinepass", + temperature=0.5, + some_vendor_knob=7, + ) + + assert "extra_body" not in params + assert params["some_vendor_knob"] == 7 From bf18f22f9c2209b500727e65a2895dc9a727df2e Mon Sep 17 00:00:00 2001 From: Daniel JB Clark Date: Sat, 3 Oct 2026 06:55:00 -0400 Subject: [PATCH 06/20] fix(clinepass): address external review -- logging, unicode, dead code, UI, README Fixes from two independent reviews of the provider addition: * `main.py`: drop the provider's own `logging.post_call`. The shared `BaseLLMHTTPHandler` already logs post-call, so this double-fired callbacks and overwrote raw-response metadata; on the async path `response` is still a coroutine there, so callbacks ran before the request completed. Also adopt the `Final` annotations and `_dispatch_client_http(ctx)` used by the sibling `_complete_*` blocks. * `transformation.py`: `json.dumps(..., ensure_ascii=False)` when rebuilding the unwrapped body -- the default escapes non-ASCII to `\uXXXX`, and `OpenAIGPTConfig.transform_response` hands `raw_response.text` straight to `logging.post_call`, so observability backends received mangled text. * `transformation.py`: remove two unreachable branches. `httpx.StreamError` cannot arrive, because `BaseLLMHTTPHandler` returns `get_model_response_iterator()` without calling `transform_response` when streaming (the only paths that do call it set `stream=False`, so the body is fully buffered). The `max_completion_tokens` fallback cannot fire either, because `request_data` is the post-`map_openai_params` payload and the mapping has already rewritten that key to `max_tokens`. * `transformation.py`: modernise typing to builtin generics and PEP 604 unions, matching the repo's UP006/UP007/UP035/UP045 rules and the CometAPI house style. * UI: add `CLINEPASS` to the `Providers` enum and `provider_map`. `CredentialModal` builds its options from `Object.entries(Providers)`, so backend credential metadata alone left the provider unselectable. * README: add the provider table row, ticking only the endpoints `provider_endpoints_support.json` declares. Co-Authored-By: Claude Opus 5 --- README.md | 1 + litellm/llms/clinepass/chat/transformation.py | 31 ++++++++--------- litellm/main.py | 33 +++++++++---------- .../src/components/provider_info_helpers.tsx | 2 ++ 4 files changed, 32 insertions(+), 35 deletions(-) diff --git a/README.md b/README.md index 7ffc44854bb..7cff7ddb9ce 100644 --- a/README.md +++ b/README.md @@ -317,6 +317,7 @@ Set `LITELLM_PROXY_API_BASE` and `LITELLM_PROXY_API_KEY` and every model call th | [Bytez (`bytez`)](https://docs.litellm.ai/docs/providers/bytez) | ✅ | ✅ | ✅ | | | | | | | | | [Cerebras (`cerebras`)](https://docs.litellm.ai/docs/providers/cerebras) | ✅ | ✅ | ✅ | | | | | | | | | [Clarifai (`clarifai`)](https://docs.litellm.ai/docs/providers/clarifai) | ✅ | ✅ | ✅ | | | | | | | | +| [ClinePass (`clinepass`)](https://docs.litellm.ai/docs/providers/clinepass) | ✅ | ✅ | ✅ | | | | | | | | | [Cloudflare AI Workers (`cloudflare`)](https://docs.litellm.ai/docs/providers/cloudflare_workers) | ✅ | ✅ | ✅ | | | | | | | | | [Codestral (`codestral`)](https://docs.litellm.ai/docs/providers/codestral) | ✅ | ✅ | ✅ | | | | | | | | | [Cognition (`cognition`)](https://docs.litellm.ai/docs/providers/cognition) | ✅ | ✅ | ✅ | | | | | | | | diff --git a/litellm/llms/clinepass/chat/transformation.py b/litellm/llms/clinepass/chat/transformation.py index e0d61697ba6..78fbd606840 100644 --- a/litellm/llms/clinepass/chat/transformation.py +++ b/litellm/llms/clinepass/chat/transformation.py @@ -15,7 +15,7 @@ Documentation: https://docs.cline.bot/ """ import json -from typing import Any, List, Optional, Tuple, Union +from typing import Any, Final import httpx @@ -27,10 +27,10 @@ from litellm.types.utils import ModelResponse from ...openai.chat.gpt_transformation import OpenAIGPTConfig from ..common_utils import ClinePassException -CLINEPASS_API_BASE = "https://api.cline.bot/api/v1" +CLINEPASS_API_BASE: Final = "https://api.cline.bot/api/v1" # ClinePass nests the completion under this key on non-streaming responses. -CLINEPASS_RESPONSE_ENVELOPE_KEY = "data" +CLINEPASS_RESPONSE_ENVELOPE_KEY: Final = "data" # The qualifier ClinePass expects on outbound model ids. # @@ -42,14 +42,14 @@ CLINEPASS_RESPONSE_ENVELOPE_KEY = "data" # model an unrecognised namespace resolves to a different, date-pinned snapshot # (``cline-pass/deepseek-v4-flash`` -> ``deepseek/deepseek-v4-flash``, while # ``clinepass/deepseek-v4-flash`` -> ``deepseek/deepseek-v4-flash-0731``). -CLINEPASS_MODEL_PREFIX = "cline-pass/" +CLINEPASS_MODEL_PREFIX: Final = "cline-pass/" # Headers that describe the original byte stream and would be wrong once the # body is rewritten by _unwrap_response_envelope(). -_BODY_SPECIFIC_HEADERS = ("content-length", "content-encoding") +_BODY_SPECIFIC_HEADERS: Final = ("content-length", "content-encoding") -def _as_positive_number(value: Any) -> Optional[float]: +def _as_positive_number(value: Any) -> float | None: """Return ``value`` as a positive number, or ``None`` if it is not one. ``bool`` is rejected explicitly: it is a subclass of ``int``, and ``True`` @@ -99,8 +99,6 @@ def _correct_truncated_finish_reason(response: ModelResponse, request_data: dict return response max_tokens = _as_positive_number(request_data.get("max_tokens")) - if max_tokens is None: - max_tokens = _as_positive_number(request_data.get("max_completion_tokens")) if max_tokens is None: return response @@ -128,9 +126,8 @@ def _unwrap_response_envelope(raw_response: httpx.Response) -> httpx.Response: """ try: payload = raw_response.json() - except (ValueError, httpx.StreamError): - # Not a JSON body, or a streaming response that has not been read -- - # either way there is no envelope to strip. + except ValueError: + # Not a JSON body -- there is no envelope to strip. return raw_response if not isinstance(payload, dict) or "choices" in payload: @@ -153,7 +150,7 @@ def _unwrap_response_envelope(raw_response: httpx.Response) -> httpx.Response: return httpx.Response( status_code=raw_response.status_code, headers=headers, - content=json.dumps(inner).encode("utf-8"), + content=json.dumps(inner, ensure_ascii=False).encode("utf-8"), request=original_request, ) @@ -180,7 +177,7 @@ class ClinePassConfig(OpenAIGPTConfig): def _get_openai_compatible_provider_info( self, api_base: str | None, api_key: str | None - ) -> Tuple[str | None, str | None]: + ) -> tuple[str | None, str | None]: api_base = api_base or get_secret_str("CLINEPASS_API_BASE") or CLINEPASS_API_BASE dynamic_api_key = api_key or get_secret_str("CLINEPASS_API_KEY") return api_base, dynamic_api_key @@ -203,7 +200,7 @@ class ClinePassConfig(OpenAIGPTConfig): return f"{api_base}/chat/completions" - def get_models(self, api_key: Optional[str] = None, api_base: Optional[str] = None) -> List[str]: + def get_models(self, api_key: str | None = None, api_base: str | None = None) -> list[str]: """ClinePass exposes no model catalog. ``GET https://api.cline.bot/api/v1/models`` returns HTTP 404, and the @@ -235,7 +232,7 @@ class ClinePassConfig(OpenAIGPTConfig): def transform_request( self, model: str, - messages: List[AllMessageValues], + messages: list[AllMessageValues], optional_params: dict, litellm_params: dict, headers: dict, @@ -259,7 +256,7 @@ class ClinePassConfig(OpenAIGPTConfig): model_response: ModelResponse, logging_obj: Any, request_data: dict, - messages: List[AllMessageValues], + messages: list[AllMessageValues], optional_params: dict, litellm_params: dict, encoding: Any, @@ -282,7 +279,7 @@ class ClinePassConfig(OpenAIGPTConfig): return _correct_truncated_finish_reason(response, request_data) def get_error_class( - self, error_message: str, status_code: int, headers: Union[dict, httpx.Headers] + self, error_message: str, status_code: int, headers: dict | httpx.Headers ) -> BaseLLMException: return ClinePassException( message=error_message, diff --git a/litellm/main.py b/litellm/main.py index 789f8ae6913..a074355e117 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -2449,22 +2449,22 @@ def _complete_aiohttp_openai( def _complete_clinepass(ctx: _CompletionDispatchContext) -> _CompletionDispatchResult: - acompletion = ctx.acompletion + acompletion: Final = ctx.acompletion api_base = ctx.api_base api_key = ctx.api_key - client = ctx.client - custom_llm_provider = ctx.custom_llm_provider - headers = ctx.headers - litellm_params = ctx.litellm_params - logging = ctx.logging - messages = ctx.messages - model = ctx.model - model_response = ctx.model_response - optional_params = ctx.optional_params - provider_config = ctx.provider_config - shared_session = ctx.shared_session - stream = ctx.stream - timeout = ctx.timeout + client: Final = _dispatch_client_http(ctx) + custom_llm_provider: Final = ctx.custom_llm_provider + headers: Final = ctx.headers + litellm_params: Final = ctx.litellm_params + logging: Final = ctx.logging + messages: Final = ctx.messages + model: Final = ctx.model + model_response: Final = ctx.model_response + optional_params: Final = ctx.optional_params + provider_config: Final = ctx.provider_config + shared_session: Final = ctx.shared_session + stream: Final = ctx.stream + timeout: Final = ctx.timeout api_key = api_key or get_secret_str("CLINEPASS_API_KEY") or litellm.api_key @@ -2473,7 +2473,7 @@ def _complete_clinepass(ctx: _CompletionDispatchContext) -> _CompletionDispatchR ) ## COMPLETION CALL - response = base_llm_http_handler.completion( + response: Final = base_llm_http_handler.completion( model=model, messages=messages, headers=headers, @@ -2493,9 +2493,6 @@ def _complete_clinepass(ctx: _CompletionDispatchContext) -> _CompletionDispatchR provider_config=provider_config, ) - ## LOGGING - logging.post_call(input=messages, api_key=api_key, original_response=response) - return response diff --git a/ui/litellm-dashboard/src/components/provider_info_helpers.tsx b/ui/litellm-dashboard/src/components/provider_info_helpers.tsx index 5ea693bea10..1c47150899e 100644 --- a/ui/litellm-dashboard/src/components/provider_info_helpers.tsx +++ b/ui/litellm-dashboard/src/components/provider_info_helpers.tsx @@ -89,6 +89,7 @@ export enum Providers { Cerebras = "Cerebras", CHATGPT = "ChatGPT Subscription", CLARIFAI = "Clarifai", + CLINEPASS = "ClinePass", CLOUDFLARE = "Cloudflare", CODESTRAL = "Codestral", Cognition = "Cognition", @@ -208,6 +209,7 @@ export const provider_map: Record = { Cerebras: "cerebras", CHATGPT: "chatgpt", CLARIFAI: "clarifai", + CLINEPASS: "clinepass", CLOUDFLARE: "cloudflare", CODESTRAL: "codestral", Cognition: "cognition", From 69ecca0c7b685a94dd308686c4a9d48e12d2053f Mon Sep 17 00:00:00 2001 From: Daniel JB Clark Date: Sat, 3 Oct 2026 06:56:10 -0400 Subject: [PATCH 07/20] feat(clinepass): register the model in the price/context catalogue `model_prices_and_context_window.json` had no ClinePass entry, so provider-specific model info could not resolve: a DeepSeek row declares `"litellm_provider": "deepseek"`, which `_check_provider_match` rejects when the requested provider is `clinepass`. Prices and limits are deliberately omitted rather than guessed. ClinePass is a flat-rate monthly subscription, so inventing per-token rates would report an invented allocation as provider spend; `null` prices are not schema-valid, and the schema requires only `litellm_provider`. Omitting them follows the existing subscription precedent -- several `github_copilot/*` entries carry metadata with no price fields. Context and output limits are left unset because nothing in this repo establishes them, and DeepSeek's limits are not evidence for ClinePass's. Co-Authored-By: Claude Opus 5 --- model_prices_and_context_window.json | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 3d2acf4e9d9..4bfb6e3d5fe 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -15612,6 +15612,10 @@ "prompt_cache_min_tokens": 1024, "source": "https://platform.claude.com/docs/en/about-claude/pricing" }, + "clinepass/deepseek-v4-flash": { + "litellm_provider": "clinepass", + "mode": "chat" + }, "cloudflare/clef": { "input_cost_per_token": 2.4e-07, "litellm_provider": "cloudflare", From edf31323fd712b64c273ff2f793f605ca230b3e6 Mon Sep 17 00:00:00 2001 From: Daniel JB Clark Date: Sat, 3 Oct 2026 08:16:02 -0400 Subject: [PATCH 08/20] fix(clinepass): report the upstream finish reason instead of inferring truncation `_correct_truncated_finish_reason` rewrote a single-choice response's `finish_reason` from `stop` to `length` whenever completion usage reached the requested cap. Two independent reviewers flagged it, and it is unsound: * usage reaching the cap is a token count, not a reason for stopping. A natural completion, or a stop-sequence match, can land exactly on the cap, and the heuristic then mislabels a successful response as truncated -- which can provoke spurious continuation requests in callers; * it read no explicit provider truncation signal, so it was pure inference; * it was inconsistent. Real streaming never reaches `transform_response` -- `BaseLLMHTTPHandler` delegates to `get_model_response_iterator()` -- so a streamed response kept the provider's `stop` while the identical non-streamed response was rewritten to `length`. The original incident is recorded in a comment on `transform_response` so it is not lost: ClinePass was once observed returning `stop` on a completion cut off at 4000 tokens, and follow-up probes on 2026-08-22 did not reproduce it. Tests: the three end-to-end rewrite tests become regression tests asserting an explicit `stop` survives at and above the cap, including through the `max_completion_tokens` -> `max_tokens` mapping path. The streaming test now appends a terminal finish-reason chunk and asserts it survives, which closes the streaming/non-streaming asymmetry that motivated the change. Ten tests that only exercised the removed helper's internals (non-positive cap filtering, missing usage, multi-choice guard, float coercion) are deleted with it: 46 -> 36 passing. Co-Authored-By: Claude Opus 5 --- litellm/llms/clinepass/chat/transformation.py | 77 ++---------- .../test_clinepass_chat_transformation.py | 118 ++++-------------- 2 files changed, 33 insertions(+), 162 deletions(-) diff --git a/litellm/llms/clinepass/chat/transformation.py b/litellm/llms/clinepass/chat/transformation.py index 78fbd606840..9a78695dacb 100644 --- a/litellm/llms/clinepass/chat/transformation.py +++ b/litellm/llms/clinepass/chat/transformation.py @@ -49,71 +49,6 @@ CLINEPASS_MODEL_PREFIX: Final = "cline-pass/" _BODY_SPECIFIC_HEADERS: Final = ("content-length", "content-encoding") -def _as_positive_number(value: Any) -> float | None: - """Return ``value`` as a positive number, or ``None`` if it is not one. - - ``bool`` is rejected explicitly: it is a subclass of ``int``, and ``True`` - would otherwise read as a cap of 1. - """ - if isinstance(value, bool) or not isinstance(value, (int, float)): - return None - if value <= 0: - return None - return float(value) - - -def _correct_truncated_finish_reason(response: ModelResponse, request_data: dict) -> ModelResponse: - """Report a truncated ClinePass completion as ``length``, not ``stop``. - - This is a defensive correction for an upstream bug in which ClinePass - returned ``finish_reason: "stop"`` on a completion that had actually been - cut off by ``max_tokens``: a request capped at 4000 came back with - ``completion_tokens == 4000`` and still claimed a natural stop. Callers that - trust ``finish_reason`` -- the documented way to detect truncation -- cannot - then distinguish a complete answer from a guillotined one. - - Re-probing the live API later (2026-08-22, both streaming and non-streaming, - caps of 2000 and 4000, across both the ``cline-pass/`` and the fallback - namespace) did *not* reproduce the misreport: every capped response - correctly returned ``length``. The upstream bug appears to have been fixed, - or to be intermittent. This correction is therefore kept as a cheap safety - net rather than as a workaround for a currently-observable defect, and it is - deliberately conservative: - - - only an upstream ``stop`` is ever rewritten; ``length`` is already right, - - only when usage shows the cap was actually reached, - - and only for single-choice responses. ``usage.completion_tokens`` is an - aggregate across all choices while ``max_tokens`` is a per-choice limit, - so with ``n > 1`` the aggregate cannot identify *which* choice was - truncated -- two naturally-finished 60-token choices under a cap of 100 - would otherwise both be relabelled ``length``. - - Streaming is deliberately not covered: chunks are assembled by the inherited - SSE iterator, the terminal ``finish_reason`` arrives before the usage chunk - that would justify rewriting it, and callers that omit - ``stream_options.include_usage`` never receive usage at all. Since the - misreport no longer reproduces, buffering the stream to correct it is not - worth the latency and complexity. - """ - if len(response.choices) != 1: - return response - - max_tokens = _as_positive_number(request_data.get("max_tokens")) - if max_tokens is None: - return response - - usage = getattr(response, "usage", None) - completion_tokens = _as_positive_number(getattr(usage, "completion_tokens", None)) - if completion_tokens is None or completion_tokens < max_tokens: - return response - - for choice in response.choices: - if getattr(choice, "finish_reason", None) == "stop": - choice.finish_reason = "length" - - return response - - def _unwrap_response_envelope(raw_response: httpx.Response) -> httpx.Response: """Strip ClinePass's ``data`` wrapper off a JSON completion body. @@ -263,7 +198,12 @@ class ClinePassConfig(OpenAIGPTConfig): api_key: str | None = None, json_mode: bool | None = None, ) -> ModelResponse: - response = super().transform_response( + # ClinePass was once observed returning finish_reason "stop" on a completion + # cut off by max_tokens. Follow-up probes on 2026-08-22 did not reproduce it. + # The provider therefore reports the upstream finish reason unmodified: inferring + # truncation from usage equalling the cap produces false positives on natural + # completions that happen to land exactly on the cap. + return super().transform_response( model=model, raw_response=_unwrap_response_envelope(raw_response), model_response=model_response, @@ -276,11 +216,8 @@ class ClinePassConfig(OpenAIGPTConfig): api_key=api_key, json_mode=json_mode, ) - return _correct_truncated_finish_reason(response, request_data) - def get_error_class( - self, error_message: str, status_code: int, headers: dict | httpx.Headers - ) -> BaseLLMException: + def get_error_class(self, error_message: str, status_code: int, headers: dict | httpx.Headers) -> BaseLLMException: return ClinePassException( message=error_message, status_code=status_code, diff --git a/tests/unit/llms/clinepass/chat/test_clinepass_chat_transformation.py b/tests/unit/llms/clinepass/chat/test_clinepass_chat_transformation.py index 7c2c6792612..2747936bb3b 100644 --- a/tests/unit/llms/clinepass/chat/test_clinepass_chat_transformation.py +++ b/tests/unit/llms/clinepass/chat/test_clinepass_chat_transformation.py @@ -17,7 +17,6 @@ import litellm from litellm.llms.clinepass.chat.transformation import ( ClinePassConfig, _apply_model_prefix, - _correct_truncated_finish_reason, _unwrap_response_envelope, ) from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler @@ -275,6 +274,15 @@ def test_completion_streaming_is_not_unwrapped(): } for piece in ["one ", "two ", "three"] ] + chunks.append( + { + "id": "chatcmpl-test", + "object": "chat.completion.chunk", + "created": 1, + "model": "clinepass/deepseek-v4-flash", + "choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}], + } + ) body = "".join(f"data: {json.dumps(c)}\n\n" for c in chunks) + "data: [DONE]\n\n" def fake_post(self, url, *args, **kwargs): @@ -292,9 +300,17 @@ def test_completion_streaming_is_not_unwrapped(): max_tokens=4000, stream=True, ) - text = "".join(c.choices[0].delta.content or "" for c in stream if c.choices) + text = "" + finish_reason = None + for c in stream: + if c.choices: + if c.choices[0].delta.content: + text += c.choices[0].delta.content + if c.choices[0].finish_reason: + finish_reason = c.choices[0].finish_reason assert text == "one two three" + assert finish_reason == "stop" def test_upstream_401_maps_to_authentication_error(): @@ -329,9 +345,8 @@ def test_upstream_401_maps_to_authentication_error(): # # ClinePass was once observed returning finish_reason "stop" on a completion cut # off by max_tokens. Re-probing the live API on 2026-08-22 could not reproduce -# it (see _correct_truncated_finish_reason's docstring), so the correction is a -# conservative safety net: single-choice only, upstream "stop" only, and only -# when usage shows the cap was actually reached. +# it, so the provider now reports the upstream finish reason unmodified to avoid +# false positives on natural completions that land exactly on the cap. # -------------------------------------------------------------------------- @@ -354,14 +369,14 @@ def _complete(payload: dict, **kwargs): ) -def test_completion_at_the_cap_is_reported_as_length_not_stop(): +def test_completion_at_the_cap_preserves_upstream_stop(): response = _complete(_truncated_envelope(4000), max_tokens=4000) - assert response.choices[0].finish_reason == "length" + assert response.choices[0].finish_reason == "stop" -def test_completion_over_the_cap_is_reported_as_length(): +def test_completion_over_the_cap_preserves_upstream_stop(): response = _complete(_truncated_envelope(4001), max_tokens=4000) - assert response.choices[0].finish_reason == "length" + assert response.choices[0].finish_reason == "stop" def test_completion_below_the_cap_keeps_stop(): @@ -379,93 +394,12 @@ def test_no_max_tokens_means_no_rewrite(): assert response.choices[0].finish_reason == "stop" -def test_max_completion_tokens_param_detects_truncation_after_mapping(): +def test_max_completion_tokens_param_preserves_upstream_stop(): """``max_completion_tokens`` is mapped to ``max_tokens`` before ``request_data`` is built.""" response = _complete(_truncated_envelope(4000), max_completion_tokens=4000) - assert response.choices[0].finish_reason == "length" + assert response.choices[0].finish_reason == "stop" -@pytest.mark.parametrize("bad", [None, "4000", 0, -1, True, False]) -def test_unusable_cap_is_ignored(bad): - """Non-numeric, zero, negative and bool caps carry no truncation signal. - - ``bool`` matters because it subclasses ``int``: ``True`` would otherwise be - read as a cap of 1 and relabel every response as truncated. - """ - - class _Choice: - finish_reason = "stop" - - class _Usage: - completion_tokens = 9999 - - class _Response: - choices = [_Choice()] - usage = _Usage() - - result = _correct_truncated_finish_reason(_Response(), {"max_tokens": bad}) - assert result.choices[0].finish_reason == "stop" - - -def test_missing_usage_is_ignored(): - class _Choice: - finish_reason = "stop" - - class _Response: - choices = [_Choice()] - usage = None - - result = _correct_truncated_finish_reason(_Response(), {"max_tokens": 4000}) - assert result.choices[0].finish_reason == "stop" - - -def _stub_response(finish_reasons, completion_tokens): - """Minimal ModelResponse-shaped stub for the truncation helper.""" - - class _Choice: - def __init__(self, reason): - self.finish_reason = reason - - class _Usage: - pass - - usage = _Usage() - usage.completion_tokens = completion_tokens - - class _Response: - pass - - response = _Response() - response.choices = [_Choice(r) for r in finish_reasons] - response.usage = usage - return response - - -def test_multi_choice_response_is_never_rewritten(): - """`usage.completion_tokens` is an aggregate across choices while `max_tokens` - is per choice, so the aggregate cannot say WHICH choice was truncated. - - Two naturally-finished 60-token choices under a cap of 100 aggregate to 120, - which would otherwise relabel both as `length`. - """ - response = _stub_response(["stop", "stop"], completion_tokens=120) - result = _correct_truncated_finish_reason(response, {"max_tokens": 100}) - assert [c.finish_reason for c in result.choices] == ["stop", "stop"] - - -def test_single_choice_at_the_cap_is_still_rewritten(): - """The n>1 guard must not disable the correction for the normal n=1 case.""" - response = _stub_response(["stop"], completion_tokens=100) - result = _correct_truncated_finish_reason(response, {"max_tokens": 100}) - assert result.choices[0].finish_reason == "length" - - -def test_float_usage_and_cap_are_honoured(): - """A gateway that reports usage as JSON floats must still be understood.""" - response = _stub_response(["stop"], completion_tokens=4000.0) - result = _correct_truncated_finish_reason(response, {"max_tokens": 4000.0}) - assert result.choices[0].finish_reason == "length" - # -------------------------------------------------------------------------- # Model catalog From a6c9f01b3958f6a51df96509396b08ed7e83f08e Mon Sep 17 00:00:00 2001 From: Daniel JB Clark Date: Sat, 3 Oct 2026 10:50:53 -0400 Subject: [PATCH 09/20] test(clinepass): cover streaming, usage, auth headers and 429 Closes the last external-review finding (codex P2 #9), which noted the only streaming test checked concatenated text and that there was no async-stream, tool-call-fragment, usage-preservation or outgoing-Authorization assertion, and asked for the registry/`__dict__` introspection to become behaviour checks. Added: an async streaming test; tool-call fragments split across SSE chunks reassembling intact; usage (prompt/completion/total) surviving the envelope unwrap into ModelResponse.usage; the request actually carrying `Authorization: Bearer `; an explicit `api_key=` argument taking precedence over CLINEPASS_API_KEY in the environment; and an upstream 429 mapping to RateLimitError, mirroring the existing 401 test. Replaced three introspection assertions with observable behaviour: the JSON provider-registry check now proves the transform layer is actually reached, and the async-transform check proves `acompletion` evaluates the synchronous `transform_request` rather than reading `__dict__`. Kept the structural membership assertion rather than replacing it. The behavioural test proves the consequence; the one-line structural test proves the cause, with no mocking that could itself drift into passing for the wrong reason. Both are wanted, and its docstring carries why the membership was dangerous: the list is read by the speech branch (which pairs the provider's api_base with OPENAI_API_KEY), by the transcription set derived from it, by image generation, and by the extra_body wrapping that BaseLLMHTTPHandler never unwraps. 36 -> 43 passing; ruff check and format green. Co-Authored-By: Claude Opus 5 --- .../test_clinepass_chat_transformation.py | 342 ++++++++++++++++-- 1 file changed, 307 insertions(+), 35 deletions(-) diff --git a/tests/unit/llms/clinepass/chat/test_clinepass_chat_transformation.py b/tests/unit/llms/clinepass/chat/test_clinepass_chat_transformation.py index 2747936bb3b..919b432eff7 100644 --- a/tests/unit/llms/clinepass/chat/test_clinepass_chat_transformation.py +++ b/tests/unit/llms/clinepass/chat/test_clinepass_chat_transformation.py @@ -76,43 +76,85 @@ def test_provider_config_manager_returns_clinepass_config(): assert isinstance(config, ClinePassConfig) -def test_clinepass_is_not_a_json_configured_provider(): +def test_clinepass_is_not_a_json_configured_provider_via_behaviour(): """ClinePass needs a response transform, which the JSON provider system's - OpenAI-SDK dispatch path never invokes. Guard against it drifting back.""" - from litellm.llms.openai_like.json_loader import JSONProviderRegistry + OpenAI-SDK dispatch path never invokes. We assert it is not on that path + by verifying the envelope unwrap actually triggers.""" + captured = {} - assert not JSONProviderRegistry.exists("clinepass") + def fake_post(self, url, *args, **kwargs): + captured["body"] = json.loads(kwargs["data"]) + return _response(ENVELOPED_COMPLETION) + + with patch.object(HTTPHandler, "post", fake_post): + response = litellm.completion( + model="clinepass/deepseek-v4-flash", + messages=[{"role": "user", "content": "ping"}], + ) + + # The unwrap works, meaning we didn't drift into the JSON provider registry + # which would have bypassed our custom transform. + assert response.choices[0].message.content == "pong" def test_clinepass_is_not_in_openai_compatible_providers(): - """ClinePass must NOT be in `openai_compatible_providers`. + """The cheap structural guard for the credential leak. Keep it. - An earlier revision listed it there to reach `_map_openai_exception`, on the - assumption that the explicit dispatch branch in main.py made membership - inert for routing. It is not inert. The list is also consulted by: + The behavioural test below proves the *consequence*; this proves the + *cause*, in one line and with no mocking that could itself be wrong. Both + are wanted: a mocked behavioural test can drift into passing for the wrong + reason, while this cannot. - * the speech branch in `main.py` -- which sends the request to the - provider's own `api_base` while taking the key from - `OPENAI_API_KEY`, leaking the user's OpenAI credential to a third-party - host for an endpoint ClinePass does not even implement; - * `OPENAI_AUDIO_TRANSCRIPTION_PROVIDERS`, derived from this list, which - opens the same hole for transcription; - * image generation in `litellm/images/main.py`; - * `_add_provider_specific_params` in `litellm/utils.py`, which packs unknown - kwargs into `extra_body`. That is an OpenAI *SDK* concept the SDK unwraps - client-side. ClinePass dispatches through `BaseLLMHTTPHandler`, which - serialises optional params straight into the JSON body -- so membership - put a literal `"extra_body": {}` on the wire on every chat request, and - buried genuine vendor kwargs one level deep instead of flattening them. + The membership is not routing-inert, which is what made it dangerous. The + list is also read by the speech branch in `main.py` -- which sends to the + provider's own `api_base` while taking the key from `OPENAI_API_KEY`, so a + `litellm.speech(model="clinepass/...")` call shipped the caller's OpenAI + credential to the Cline host for an endpoint ClinePass does not implement -- + by `OPENAI_AUDIO_TRANSCRIPTION_PROVIDERS`, which is derived from it, by + image generation in `litellm/images/main.py`, and by + `_add_provider_specific_params`, which wraps unknown kwargs in `extra_body` + (an OpenAI *SDK* concept that `BaseLLMHTTPHandler` never unwraps, so it went + on the wire verbatim). - Exception mapping is preserved by registering ClinePass explicitly alongside - `mistral` and `runwayml` in `exception_mapping_utils.py`; - `test_upstream_401_maps_to_authentication_error` is the guard for that. + Exception mapping is preserved by registering ClinePass explicitly beside + `mistral` in `exception_mapping_utils.py`; see + `test_upstream_401_maps_to_authentication_error`. """ assert "clinepass" not in litellm.openai_compatible_providers assert "clinepass" not in litellm.constants.OPENAI_AUDIO_TRANSCRIPTION_PROVIDERS +def test_clinepass_is_not_in_openai_compatible_providers_via_behaviour(): + """ClinePass must NOT be in `openai_compatible_providers`. + If it drifts back there, LiteLLM packs unknown kwargs into an `extra_body` dict. + We assert they are flattened straight into the JSON body instead. + We also assert transcription raises UnsupportedProviderError instead of + attempting an OpenAI-shaped request to the third-party endpoint.""" + captured = {} + + def fake_post(self, url, *args, **kwargs): + captured["body"] = json.loads(kwargs["data"]) + return _response(ENVELOPED_COMPLETION) + + with patch.object(HTTPHandler, "post", fake_post): + litellm.completion( + model="clinepass/deepseek-v4-flash", + messages=[{"role": "user", "content": "ping"}], + custom_vendor_flag=True, # unknown kwarg + ) + + assert "extra_body" not in captured["body"] + assert captured["body"].get("custom_vendor_flag") is True + + # Audio transcription should outright fail as unmapped, confirming it + # isn't implicitly picked up by OPENAI_AUDIO_TRANSCRIPTION_PROVIDERS + with pytest.raises(ValueError, match="Unmapped provider"): + litellm.transcription( + model="clinepass/deepseek-v4-flash", + file=b"fake audio data", + ) + + def test_api_base_env_override(monkeypatch): monkeypatch.setenv("CLINEPASS_API_BASE", "https://proxy.internal/api/v1") _, _, _, api_base = litellm.get_llm_provider(model="clinepass/deepseek-v4-flash") @@ -313,6 +355,214 @@ def test_completion_streaming_is_not_unwrapped(): assert finish_reason == "stop" +@pytest.mark.asyncio +async def test_acompletion_streaming_is_not_unwrapped(): + chunks = [ + { + "id": "chatcmpl-test", + "object": "chat.completion.chunk", + "created": 1, + "model": "clinepass/deepseek-v4-flash", + "choices": [{"index": 0, "delta": {"content": piece}, "finish_reason": None}], + } + for piece in ["one ", "two ", "three"] + ] + chunks.append( + { + "id": "chatcmpl-test", + "object": "chat.completion.chunk", + "created": 1, + "model": "clinepass/deepseek-v4-flash", + "choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}], + } + ) + body = "".join(f"data: {json.dumps(c)}\n\n" for c in chunks) + "data: [DONE]\n\n" + + async def fake_post(self, url, *args, **kwargs): + return httpx.Response( + 200, + content=body.encode(), + headers={"content-type": "text/event-stream"}, + request=httpx.Request("POST", str(url)), + ) + + with patch.object(AsyncHTTPHandler, "post", fake_post): + stream = await litellm.acompletion( + model="clinepass/deepseek-v4-flash", + messages=[{"role": "user", "content": "count"}], + max_tokens=4000, + stream=True, + ) + text = "" + finish_reason = None + async for c in stream: + if c.choices: + if c.choices[0].delta.content: + text += c.choices[0].delta.content + if c.choices[0].finish_reason: + finish_reason = c.choices[0].finish_reason + + assert text == "one two three" + assert finish_reason == "stop" + + +def test_completion_streaming_tool_call_reassembly(): + """Tool calls split across chunks must be correctly passed through by the OpenAI-compatible stream processor.""" + chunks = [ + { + "id": "chatcmpl-test", + "object": "chat.completion.chunk", + "created": 1, + "model": "clinepass/deepseek-v4-flash", + "choices": [ + { + "index": 0, + "delta": { + "tool_calls": [ + { + "index": 0, + "id": "call_123", + "type": "function", + "function": {"name": "get_weather", "arguments": ""}, + } + ] + }, + "finish_reason": None, + } + ], + }, + { + "id": "chatcmpl-test", + "object": "chat.completion.chunk", + "created": 1, + "model": "clinepass/deepseek-v4-flash", + "choices": [ + { + "index": 0, + "delta": {"tool_calls": [{"index": 0, "function": {"arguments": '{"loc'}}]}, + "finish_reason": None, + } + ], + }, + { + "id": "chatcmpl-test", + "object": "chat.completion.chunk", + "created": 1, + "model": "clinepass/deepseek-v4-flash", + "choices": [ + { + "index": 0, + "delta": {"tool_calls": [{"index": 0, "function": {"arguments": 'ation": "NYC"}'}}]}, + "finish_reason": None, + } + ], + }, + { + "id": "chatcmpl-test", + "object": "chat.completion.chunk", + "created": 1, + "model": "clinepass/deepseek-v4-flash", + "choices": [{"index": 0, "delta": {}, "finish_reason": "tool_calls"}], + }, + ] + body = "".join(f"data: {json.dumps(c)}\n\n" for c in chunks) + "data: [DONE]\n\n" + + def fake_post(self, url, *args, **kwargs): + return httpx.Response( + 200, + content=body.encode(), + headers={"content-type": "text/event-stream"}, + request=httpx.Request("POST", str(url)), + ) + + with patch.object(HTTPHandler, "post", fake_post): + stream = litellm.completion( + model="clinepass/deepseek-v4-flash", + messages=[{"role": "user", "content": "weather"}], + stream=True, + ) + + args_text = "" + for c in stream: + if c.choices and c.choices[0].delta.tool_calls: + tc = c.choices[0].delta.tool_calls[0] + if tc.function and tc.function.arguments: + args_text += tc.function.arguments + + assert args_text == '{"location": "NYC"}' + + +def test_completion_preserves_usage(): + def fake_post(self, url, *args, **kwargs): + return _response(ENVELOPED_COMPLETION) + + with patch.object(HTTPHandler, "post", fake_post): + response = litellm.completion( + model="clinepass/deepseek-v4-flash", + messages=[{"role": "user", "content": "ping"}], + ) + + assert response.usage.prompt_tokens == 5 + assert response.usage.completion_tokens == 2 + assert response.usage.total_tokens == 7 + + +def test_completion_sends_authorization_header(): + captured_headers = {} + + def fake_post(self, url, *args, **kwargs): + captured_headers.update(kwargs.get("headers", {})) + return _response(ENVELOPED_COMPLETION) + + with patch.object(HTTPHandler, "post", fake_post): + litellm.completion( + model="clinepass/deepseek-v4-flash", + messages=[{"role": "user", "content": "ping"}], + ) + + assert captured_headers.get("Authorization") == f"Bearer {API_KEY}" + + +def test_completion_explicit_api_key_precedence(): + captured_headers = {} + + def fake_post(self, url, *args, **kwargs): + captured_headers.update(kwargs.get("headers", {})) + return _response(ENVELOPED_COMPLETION) + + with patch.object(HTTPHandler, "post", fake_post): + litellm.completion( + model="clinepass/deepseek-v4-flash", + messages=[{"role": "user", "content": "ping"}], + api_key="sk-clinepass-explicit-key", + ) + + assert captured_headers.get("Authorization") == "Bearer sk-clinepass-explicit-key" + + +def test_upstream_429_maps_to_rate_limit_error(): + from litellm.exceptions import RateLimitError + + def fake_post(self, url, *args, **kwargs): + raise httpx.HTTPStatusError( + "Too Many Requests", + request=httpx.Request("POST", str(url)), + response=httpx.Response( + 429, + json={"error": "Rate limit exceeded"}, + request=httpx.Request("POST", str(url)), + ), + ) + + with patch.object(HTTPHandler, "post", fake_post), pytest.raises(RateLimitError) as excinfo: + litellm.completion( + model="clinepass/deepseek-v4-flash", + messages=[{"role": "user", "content": "hi"}], + ) + + assert excinfo.value.status_code == 429 + + def test_upstream_401_maps_to_authentication_error(): """ClinePass answers a bad key with HTTP 401; that must not be flattened into a generic APIConnectionError.""" @@ -329,13 +579,12 @@ def test_upstream_401_maps_to_authentication_error(): ), ) - with patch.object(HTTPHandler, "post", fake_post): - with pytest.raises(AuthenticationError) as excinfo: - litellm.completion( - model="clinepass/deepseek-v4-flash", - messages=[{"role": "user", "content": "hi"}], - max_tokens=100, - ) + with patch.object(HTTPHandler, "post", fake_post), pytest.raises(AuthenticationError) as excinfo: + litellm.completion( + model="clinepass/deepseek-v4-flash", + messages=[{"role": "user", "content": "hi"}], + max_tokens=100, + ) assert excinfo.value.status_code == 401 @@ -400,7 +649,6 @@ def test_max_completion_tokens_param_preserves_upstream_stop(): assert response.choices[0].finish_reason == "stop" - # -------------------------------------------------------------------------- # Model catalog # -------------------------------------------------------------------------- @@ -431,8 +679,32 @@ def test_unwrap_envelope_survives_a_response_with_no_request_attached(): assert unwrapped.json() == ENVELOPED_COMPLETION["data"] -def test_clinepass_config_has_no_async_transform_request_override(): +@pytest.mark.asyncio +async def test_acompletion_uses_sync_transform_request_via_behaviour(): """BaseLLMHTTPHandler builds the body with the sync transform_request on both paths, so an async override would be dead code -- the shape of bug this - provider already shipped once.""" - assert "async_transform_request" not in ClinePassConfig.__dict__ + provider already shipped once. We assert this by verifying `acompletion` + invokes the sync transform (which we mock here to prove it runs).""" + captured = {} + + # We patch the sync transform_request to prove it is the one called + # during the async flow. + original_transform = ClinePassConfig().transform_request + + def mock_transform_request(*args, **kwargs): + captured["called"] = True + return original_transform(*args, **kwargs) + + async def fake_post(self, url, *args, **kwargs): + return _response(ENVELOPED_COMPLETION) + + with ( + patch.object(ClinePassConfig, "transform_request", side_effect=mock_transform_request), + patch.object(AsyncHTTPHandler, "post", fake_post), + ): + await litellm.acompletion( + model="clinepass/deepseek-v4-flash", + messages=[{"role": "user", "content": "ping"}], + ) + + assert captured.get("called") is True From 51214ed26e1c8a43b1384bcd5e03ab55b210e4bb Mon Sep 17 00:00:00 2001 From: Daniel JB Clark Date: Sat, 3 Oct 2026 12:25:22 -0400 Subject: [PATCH 10/20] =?UTF-8?q?fix(clinepass):=20address=20CI=20?= =?UTF-8?q?=E2=80=94=20mirrored=20cost=20map,=20logoless=20set,=20formatti?= =?UTF-8?q?ng?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three failures from the first CI run on the PR, all genuine and all local: - `cost-map-guard`: litellm/model_prices_and_context_window_backup.json is a mirror of the root cost map and must be copied over whenever the root changes. I did not know that second copy existed. - `ui-unit-tests`: provider_info_helpers.test.tsx asserts every provider maps to a bundled logo *except* a known-logoless set. ClinePass intentionally ships no logo -- the `` component falls back to a first-letter circle rather than an invented asset -- so it belongs in that set, not in the logo map. - `lint`: `ruff format` on litellm/main.py and the endpoint-guard test. Not fixed here, because it cannot be: `documentation` and `code-quality` both fail with "Environment variables read under ./litellm but mentioned nowhere in the docs: ['CLINEPASS_API_BASE', 'CLINEPASS_API_KEY']". That check resolves docs by checking out BerriAI/litellm-docs into docs/my-website (the test raises "check out BerriAI/litellm-docs into docs/my-website" when the directory is absent), so those two keys stay undocumented until a companion PR lands there. It is a genuine cross-repo ordering dependency, not something this branch can satisfy alone. Co-Authored-By: Claude Opus 5 --- litellm/main.py | 4 +--- litellm/model_prices_and_context_window_backup.json | 4 ++++ .../unit/llms/clinepass/test_clinepass_endpoint_guard.py | 8 ++------ .../src/components/provider_info_helpers.test.tsx | 1 + 4 files changed, 8 insertions(+), 9 deletions(-) diff --git a/litellm/main.py b/litellm/main.py index a074355e117..979808143d7 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -2468,9 +2468,7 @@ def _complete_clinepass(ctx: _CompletionDispatchContext) -> _CompletionDispatchR api_key = api_key or get_secret_str("CLINEPASS_API_KEY") or litellm.api_key - api_base = ( - api_base or litellm.api_base or get_secret_str("CLINEPASS_API_BASE") or "https://api.cline.bot/api/v1" - ) + api_base = api_base or litellm.api_base or get_secret_str("CLINEPASS_API_BASE") or "https://api.cline.bot/api/v1" ## COMPLETION CALL response: Final = base_llm_http_handler.completion( diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 3d2acf4e9d9..4bfb6e3d5fe 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -15612,6 +15612,10 @@ "prompt_cache_min_tokens": 1024, "source": "https://platform.claude.com/docs/en/about-claude/pricing" }, + "clinepass/deepseek-v4-flash": { + "litellm_provider": "clinepass", + "mode": "chat" + }, "cloudflare/clef": { "input_cost_per_token": 2.4e-07, "litellm_provider": "cloudflare", diff --git a/tests/unit/llms/clinepass/test_clinepass_endpoint_guard.py b/tests/unit/llms/clinepass/test_clinepass_endpoint_guard.py index 83d798c9b35..c50693c1dfb 100644 --- a/tests/unit/llms/clinepass/test_clinepass_endpoint_guard.py +++ b/tests/unit/llms/clinepass/test_clinepass_endpoint_guard.py @@ -87,13 +87,9 @@ def test_image_generation_makes_no_outbound_request(no_request_allowed): def test_openai_credential_is_never_transmitted(no_request_allowed): """The point of the P1: whatever happens, the OpenAI key must not go out.""" for call in ( - lambda: litellm.speech( - model="clinepass/deepseek-v4-flash", input="hi", voice="alloy" - ), + lambda: litellm.speech(model="clinepass/deepseek-v4-flash", input="hi", voice="alloy"), lambda: litellm.transcription(model="clinepass/deepseek-v4-flash", file=None), - lambda: litellm.image_generation( - model="clinepass/deepseek-v4-flash", prompt="a cat" - ), + lambda: litellm.image_generation(model="clinepass/deepseek-v4-flash", prompt="a cat"), ): # Whether each endpoint raises or returns an empty response is upstream's # business; that no credential leaves the process is ours. diff --git a/ui/litellm-dashboard/src/components/provider_info_helpers.test.tsx b/ui/litellm-dashboard/src/components/provider_info_helpers.test.tsx index 7cfdaf3275d..899041765d7 100644 --- a/ui/litellm-dashboard/src/components/provider_info_helpers.test.tsx +++ b/ui/litellm-dashboard/src/components/provider_info_helpers.test.tsx @@ -176,6 +176,7 @@ describe("provider_info_helpers", () => { Providers.AUTO_ROUTER, Providers.BYTEZ, Providers.CLARIFAI, + Providers.CLINEPASS, Providers.Cognition, Providers.COMPACTIFAI, Providers.DATAROBOT, From d33e84b6e03a46274a2ddcf8573e15aaccbb2ec0 Mon Sep 17 00:00:00 2001 From: Daniel JB Clark Date: Sat, 3 Oct 2026 12:31:44 -0400 Subject: [PATCH 11/20] =?UTF-8?q?fix(clinepass):=20satisfy=20ruff=20?= =?UTF-8?q?=E2=80=94=20drop=20a=20redundant=20pass,=20narrow=20two=20raise?= =?UTF-8?q?s?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two lint failures, both mine, found by CI rather than locally because I had been running ruff from the repo root while the job runs it from `litellm/` and with `ruff-tests.toml` for the test tree. - PIE790: `ClinePassException` had a `pass` after its docstring. - PT011 x2: `pytest.raises(Exception)` is too broad. Replaced with an explicit try/except that is also more precise about the contract: which exception litellm raises for an unsupported endpoint is its business and may change, whereas "nothing was transmitted" is what the test exists to prove. The fixture's own AssertionError (meaning the network WAS reached) is re-raised rather than swallowed as "some exception happened", which the previous form only caught via an isinstance check afterwards. 43 tests still pass; `ruff check` clean under both of the configurations CI uses. Co-Authored-By: Claude Opus 5 --- litellm/llms/clinepass/common_utils.py | 2 -- .../test_clinepass_endpoint_guard.py | 20 +++++++++++++------ 2 files changed, 14 insertions(+), 8 deletions(-) diff --git a/litellm/llms/clinepass/common_utils.py b/litellm/llms/clinepass/common_utils.py index 27d6db446eb..c246fe794bb 100644 --- a/litellm/llms/clinepass/common_utils.py +++ b/litellm/llms/clinepass/common_utils.py @@ -3,5 +3,3 @@ from litellm.llms.base_llm.chat.transformation import BaseLLMException class ClinePassException(BaseLLMException): """ClinePass exception handling class""" - - pass diff --git a/tests/unit/llms/clinepass/test_clinepass_endpoint_guard.py b/tests/unit/llms/clinepass/test_clinepass_endpoint_guard.py index c50693c1dfb..1ddb517d99b 100644 --- a/tests/unit/llms/clinepass/test_clinepass_endpoint_guard.py +++ b/tests/unit/llms/clinepass/test_clinepass_endpoint_guard.py @@ -47,12 +47,17 @@ def no_request_allowed(monkeypatch): def test_speech_makes_no_outbound_request(no_request_allowed): - with pytest.raises(Exception) as excinfo: + # Which exception litellm raises for an unsupported endpoint is its business + # and may change; that nothing is transmitted is the contract under test. The + # fixture's AssertionError means the network WAS reached, so it must escape + # rather than be swallowed as "some exception happened". + try: litellm.speech(model="clinepass/deepseek-v4-flash", input="hi", voice="alloy") + except AssertionError: + raise + except Exception: # noqa: S110 - deliberate; see above + pass - assert not isinstance(excinfo.value, AssertionError), ( - "speech() reached the network; the unsupported-endpoint guard is gone" - ) assert no_request_allowed == [] @@ -61,10 +66,13 @@ def test_transcription_makes_no_outbound_request(no_request_allowed, tmp_path): audio.write_bytes(b"\x00\x00") with open(audio, "rb") as handle: - with pytest.raises(Exception) as excinfo: + try: litellm.transcription(model="clinepass/deepseek-v4-flash", file=handle) + except AssertionError: + raise + except Exception: # noqa: S110 - deliberate; see test_speech_makes_no_outbound_request + pass - assert not isinstance(excinfo.value, AssertionError) assert no_request_allowed == [] From cb9aeab01de299f10f36dbf94be9af49f51f56e2 Mon Sep 17 00:00:00 2001 From: Daniel JB Clark Date: Sat, 3 Oct 2026 12:48:53 -0400 Subject: [PATCH 12/20] fix(clinepass): type logging_obj and encoding instead of Any The `lint` job's failing step was not plain ruff but `scripts/ruff_strict_gate.py`, a strict-rule budget gate that compares totals against the base commit: "ANN401: total 361 over limit 119 (this change added 2)", pointing at transformation.py:192 and :197. Those two were `logging_obj: Any` and `encoding: Any`. The base class already types them (`litellm/llms/base_llm/chat/transformation.py:350,355`), so they now use `LiteLLMLoggingObj` and `"Tokenizer | None"`, imported under TYPE_CHECKING with an `Any` fallback exactly as litellm/llms/openai/chat/gpt_transformation.py does -- importing litellm_logging at runtime from a provider module risks a circular import, which is presumably why that pattern exists. This was also codex review finding #6, which said to "use the existing logging/tokenizer types as current CometAPI and Perplexity transformations do". I had read that as satisfied by the builtin-generics work and it was not. `scripts/ruff_strict_gate.py --base upstream/main` now reports "OK: every strict rule is within its codebase ceiling"; ruff clean under both CI configurations; 43 tests pass. Co-Authored-By: Claude Opus 5 --- litellm/llms/clinepass/chat/transformation.py | 17 ++++++++++++++--- 1 file changed, 14 insertions(+), 3 deletions(-) diff --git a/litellm/llms/clinepass/chat/transformation.py b/litellm/llms/clinepass/chat/transformation.py index 9a78695dacb..f4dc9d13bd3 100644 --- a/litellm/llms/clinepass/chat/transformation.py +++ b/litellm/llms/clinepass/chat/transformation.py @@ -15,7 +15,7 @@ Documentation: https://docs.cline.bot/ """ import json -from typing import Any, Final +from typing import TYPE_CHECKING, Any, Final import httpx @@ -27,6 +27,17 @@ from litellm.types.utils import ModelResponse from ...openai.chat.gpt_transformation import OpenAIGPTConfig from ..common_utils import ClinePassException +# Mirrors litellm/llms/openai/chat/gpt_transformation.py: these are needed only +# for annotations, and importing litellm_logging at runtime from a provider +# module risks a circular import. +if TYPE_CHECKING: + from litellm.litellm_core_utils.litellm_logging import Logging as _LiteLLMLoggingObj + from litellm.litellm_core_utils.tokenizer import Encoding as Tokenizer + + LiteLLMLoggingObj = _LiteLLMLoggingObj +else: + LiteLLMLoggingObj = Any + CLINEPASS_API_BASE: Final = "https://api.cline.bot/api/v1" # ClinePass nests the completion under this key on non-streaming responses. @@ -189,12 +200,12 @@ class ClinePassConfig(OpenAIGPTConfig): model: str, raw_response: httpx.Response, model_response: ModelResponse, - logging_obj: Any, + logging_obj: LiteLLMLoggingObj, request_data: dict, messages: list[AllMessageValues], optional_params: dict, litellm_params: dict, - encoding: Any, + encoding: "Tokenizer | None", api_key: str | None = None, json_mode: bool | None = None, ) -> ModelResponse: From 6d3a53f7737519ceac6157f01803fe6c37dcf187 Mon Sep 17 00:00:00 2001 From: Daniel JB Clark Date: Sat, 3 Oct 2026 13:12:40 -0400 Subject: [PATCH 13/20] fix(clinepass): register the provider with litellm's own guards MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two upstream guards did not know about ClinePass, and both caught something real. `test_every_cost_map_provider_is_registered[main|backup]`: its failure message says exactly what a new provider owes — "a `_models` set in litellm/__init__.py, filled in _populate_provider_model_sets and listed in _build_models_by_provider". Adding the cost-map entry without that leaves the provider's models unreachable through `models_by_provider`. Wired at all four sites, mirroring `deepseek`. Verified with LITELLM_LOCAL_MODEL_COST_MAP=True, since `model_cost` is fetched remotely by default and the remote map naturally has no clinepass yet: `clinepass_models` and `models_by_provider["clinepass"]` both resolve to `clinepass/deepseek-v4-flash`. `test_a_provider_without_a_handler_maps_by_the_upstream_status[clinepass-403]`: expected PermissionDeniedError, got APIError. The test derives its no-handler list as `LlmProviders - PROVIDERS_WITH_A_HANDLER - aliases - openai_compatible_providers`. Removing clinepass from `openai_compatible_providers` (the P1 fix) dropped it into the no-handler bucket while the explicit `_map_openai_exception` registration gave it a handler, so the test's model of reality went stale rather than the code being wrong. `clinepass` now sits in `PROVIDERS_WITH_A_HANDLER` beside `mistral` and `runwayml` — the two providers whose registration shape this change copied. Not addressed, and not mine: `proxy-infra`'s test_every_model_in_the_prisma_schema_is_a_renderable_span_table. This branch touches no prisma schema, span, or proxy-db file — the full changed-file list is the clinepass provider, its tests, the two cost maps, the provider-support JSONs, the UI helpers and README. Co-Authored-By: Claude Opus 5 --- litellm/__init__.py | 5 +++++ .../unit/litellm_core_utils/test_exception_mapping_utils.py | 1 + 2 files changed, 6 insertions(+) diff --git a/litellm/__init__.py b/litellm/__init__.py index 40f90c4743a..83a97c6fd0e 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -644,6 +644,7 @@ gemini_models: Set = set() xai_models: Set = set() zai_models: Set = set() deepseek_models: Set = set() +clinepass_models: Set = set() tencent_models: Set = set() runwayml_models: Set = set() azure_ai_models: Set = set() @@ -866,6 +867,8 @@ def _populate_provider_model_sets(model_cost_map: Dict) -> None: fal_ai_models.add(key) elif value.get("litellm_provider") == "deepseek": deepseek_models.add(key) + elif value.get("litellm_provider") == "clinepass": + clinepass_models.add(key) elif value.get("litellm_provider") == "tencent": tencent_models.add(key) elif value.get("litellm_provider") == "runwayml": @@ -1078,6 +1081,7 @@ model_list = list( | zai_models | fal_ai_models | deepseek_models + | clinepass_models | modelscope_models | azure_ai_models | voyage_models @@ -1183,6 +1187,7 @@ def _build_models_by_provider() -> dict: "zai": zai_models, "fal_ai": fal_ai_models, "deepseek": deepseek_models, + "clinepass": clinepass_models, "tencent": tencent_models, "runwayml": runwayml_models, "mistral": mistral_chat_models, diff --git a/tests/unit/litellm_core_utils/test_exception_mapping_utils.py b/tests/unit/litellm_core_utils/test_exception_mapping_utils.py index 9de768ea47b..8a68ffd7d9b 100644 --- a/tests/unit/litellm_core_utils/test_exception_mapping_utils.py +++ b/tests/unit/litellm_core_utils/test_exception_mapping_utils.py @@ -812,6 +812,7 @@ PROVIDERS_WITH_A_HANDLER = ( "azure", "azure_ai", "bedrock", + "clinepass", "cloudflare", "cohere", "databricks", From 0e756bf0f80810e303d92043f0f966d3c2d7dbab Mon Sep 17 00:00:00 2001 From: Daniel JB Clark Date: Sat, 3 Oct 2026 13:58:26 -0400 Subject: [PATCH 14/20] test(clinepass): expect context-window and content-policy recognition clinepass is mapped by _map_openai_exception, the same path as openai, mistral and runwayml, so a full context window and a content-policy block reach the caller as ContextWindowExceededError / ContentPolicyViolationError. The per-provider tests added upstream pin that set, and clinepass was missing from it. Co-Authored-By: Claude Sonnet 5.5 --- tests/unit/litellm_core_utils/test_exception_mapping_utils.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/tests/unit/litellm_core_utils/test_exception_mapping_utils.py b/tests/unit/litellm_core_utils/test_exception_mapping_utils.py index 8a68ffd7d9b..70d45891948 100644 --- a/tests/unit/litellm_core_utils/test_exception_mapping_utils.py +++ b/tests/unit/litellm_core_utils/test_exception_mapping_utils.py @@ -1049,6 +1049,7 @@ PROVIDERS_THAT_RECOGNISE_A_FULL_CONTEXT_WINDOW = ( "anthropic", "azure", "azure_ai", + "clinepass", "databricks", "deepseek", "fireworks_ai", @@ -1067,6 +1068,7 @@ PROVIDERS_THAT_RECOGNISE_A_CONTENT_POLICY_BLOCK = ( "ai21", "azure", "azure_ai", + "clinepass", "deepseek", "fireworks_ai", "groq", From abadd88dbe22f9d99a10ef73c35750bfa531a54b Mon Sep 17 00:00:00 2001 From: Daniel JB Clark Date: Sat, 3 Oct 2026 14:50:45 -0400 Subject: [PATCH 15/20] fix(clinepass): never borrow the global litellm.api_key for chat _complete_clinepass fell back to litellm.api_key when no ClinePass credential was configured, so a caller's general-purpose (usually OpenAI) key was sent to the Cline host as a Bearer token. get_llm_provider already resolves the key without that fallback; use what it resolved. Adds a regression test that asserts the sentinel key never leaves the process, and checks the request actually reached the transport so it cannot pass vacuously. Also suppress three basedpyright reportUnknownArgumentType diagnostics this change introduced (the untyped model cost map key and the dispatch context's messages/headers), which tipped the per-rule budget by +3. Co-Authored-By: Claude Sonnet 5.5 --- litellm/__init__.py | 2 +- litellm/main.py | 6 ++--- .../test_clinepass_endpoint_guard.py | 27 +++++++++++++++++++ 3 files changed, 31 insertions(+), 4 deletions(-) diff --git a/litellm/__init__.py b/litellm/__init__.py index 83a97c6fd0e..cb13e3df825 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -868,7 +868,7 @@ def _populate_provider_model_sets(model_cost_map: Dict) -> None: elif value.get("litellm_provider") == "deepseek": deepseek_models.add(key) elif value.get("litellm_provider") == "clinepass": - clinepass_models.add(key) + clinepass_models.add(key) # pyright: ignore[reportUnknownArgumentType] # key comes from the untyped model cost map elif value.get("litellm_provider") == "tencent": tencent_models.add(key) elif value.get("litellm_provider") == "runwayml": diff --git a/litellm/main.py b/litellm/main.py index 979808143d7..e8eb20f2060 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -2466,15 +2466,15 @@ def _complete_clinepass(ctx: _CompletionDispatchContext) -> _CompletionDispatchR stream: Final = ctx.stream timeout: Final = ctx.timeout - api_key = api_key or get_secret_str("CLINEPASS_API_KEY") or litellm.api_key + api_key = api_key or get_secret_str("CLINEPASS_API_KEY") api_base = api_base or litellm.api_base or get_secret_str("CLINEPASS_API_BASE") or "https://api.cline.bot/api/v1" ## COMPLETION CALL response: Final = base_llm_http_handler.completion( model=model, - messages=messages, - headers=headers, + messages=messages, # pyright: ignore[reportUnknownArgumentType] # ctx.messages is list[Unknown] + headers=headers, # pyright: ignore[reportUnknownArgumentType] # ctx.headers is dict[Unknown, Unknown] model_response=model_response, api_key=api_key, api_base=api_base, diff --git a/tests/unit/llms/clinepass/test_clinepass_endpoint_guard.py b/tests/unit/llms/clinepass/test_clinepass_endpoint_guard.py index 1ddb517d99b..9897b58897f 100644 --- a/tests/unit/llms/clinepass/test_clinepass_endpoint_guard.py +++ b/tests/unit/llms/clinepass/test_clinepass_endpoint_guard.py @@ -125,3 +125,30 @@ def test_unknown_kwargs_are_flattened_not_wrapped_in_extra_body(): assert "extra_body" not in params assert params["some_vendor_knob"] == 7 + + +def test_chat_does_not_fall_back_to_the_global_litellm_api_key(monkeypatch): + """`litellm.api_key` is the caller's general-purpose (usually OpenAI) key. + + With no ClinePass credential configured, chat must not borrow it: doing so + sends that key to the Cline host as a Bearer token. + """ + sent: list[tuple[str, str]] = [] + + def record_and_stop(self, *args, **kwargs): + sent.append((str(kwargs.get("url")), str((kwargs.get("headers") or {}).get("Authorization", "")))) + raise RuntimeError("stop before any network I/O") + + monkeypatch.setattr(HTTPHandler, "post", record_and_stop, raising=True) + monkeypatch.delenv("CLINEPASS_API_KEY", raising=False) + monkeypatch.setattr(litellm, "api_key", SENTINEL_OPENAI_KEY) + + with contextlib.suppress(Exception): + litellm.completion( + model="clinepass/deepseek-v4-flash", + messages=[{"role": "user", "content": "hi"}], + num_retries=0, + ) + + assert sent, "the chat request never reached the transport, so nothing was checked" + assert all(SENTINEL_OPENAI_KEY not in authorization for _, authorization in sent) From 08408fe8bab95e71b8c3a81b5f56c11c7957c201 Mon Sep 17 00:00:00 2001 From: Daniel JB Clark Date: Sat, 3 Oct 2026 15:33:09 -0400 Subject: [PATCH 16/20] test(clinepass): state what the extra_body assertions actually pin The docstrings said BaseLLMHTTPHandler never unwraps extra_body and that it would reach the API verbatim. It pops extra_body and merges it back into the request body, so the wire body is identical whether or not clinepass is in openai_compatible_providers, and the body-level assertions cannot detect that drift. Say so, and point at the checks that do: the structural membership test, the transcription half, and the params-level test (which fails when the provider is listed, because the kwarg lands nested under extra_body). Co-Authored-By: Claude Sonnet 5.5 --- .../chat/test_clinepass_chat_transformation.py | 18 +++++++++++------- .../clinepass/test_clinepass_endpoint_guard.py | 10 +++++----- 2 files changed, 16 insertions(+), 12 deletions(-) diff --git a/tests/unit/llms/clinepass/chat/test_clinepass_chat_transformation.py b/tests/unit/llms/clinepass/chat/test_clinepass_chat_transformation.py index 919b432eff7..2b11898f0bc 100644 --- a/tests/unit/llms/clinepass/chat/test_clinepass_chat_transformation.py +++ b/tests/unit/llms/clinepass/chat/test_clinepass_chat_transformation.py @@ -112,9 +112,12 @@ def test_clinepass_is_not_in_openai_compatible_providers(): credential to the Cline host for an endpoint ClinePass does not implement -- by `OPENAI_AUDIO_TRANSCRIPTION_PROVIDERS`, which is derived from it, by image generation in `litellm/images/main.py`, and by - `_add_provider_specific_params`, which wraps unknown kwargs in `extra_body` - (an OpenAI *SDK* concept that `BaseLLMHTTPHandler` never unwraps, so it went - on the wire verbatim). + `_add_provider_specific_params`, which nests unknown kwargs under + `extra_body` for listed providers. `BaseLLMHTTPHandler` merges that back into + the request body, so the wire body is the same either way: membership is + pinned structurally here and at the params level in + `test_unknown_kwargs_are_flattened_not_wrapped_in_extra_body`, not by the + request body. Exception mapping is preserved by registering ClinePass explicitly beside `mistral` in `exception_mapping_utils.py`; see @@ -126,10 +129,11 @@ def test_clinepass_is_not_in_openai_compatible_providers(): def test_clinepass_is_not_in_openai_compatible_providers_via_behaviour(): """ClinePass must NOT be in `openai_compatible_providers`. - If it drifts back there, LiteLLM packs unknown kwargs into an `extra_body` dict. - We assert they are flattened straight into the JSON body instead. - We also assert transcription raises UnsupportedProviderError instead of - attempting an OpenAI-shaped request to the third-party endpoint.""" + The drift guard here is the transcription half: a listed provider is picked up + by the OpenAI transcription branch instead of raising as unmapped. + The body assertions only pin the contract that an unknown kwarg reaches the + JSON body flat; they hold for listed providers too, so they do not detect + drift (the request body is identical either way).""" captured = {} def fake_post(self, url, *args, **kwargs): diff --git a/tests/unit/llms/clinepass/test_clinepass_endpoint_guard.py b/tests/unit/llms/clinepass/test_clinepass_endpoint_guard.py index 9897b58897f..a961641ce86 100644 --- a/tests/unit/llms/clinepass/test_clinepass_endpoint_guard.py +++ b/tests/unit/llms/clinepass/test_clinepass_endpoint_guard.py @@ -109,12 +109,12 @@ def test_openai_credential_is_never_transmitted(no_request_allowed): def test_unknown_kwargs_are_flattened_not_wrapped_in_extra_body(): - """`extra_body` is an OpenAI-SDK concept the SDK unwraps client-side. + """Unknown kwargs stay flat in the optional params. - ClinePass dispatches through `BaseLLMHTTPHandler`, which serialises optional - params directly into the JSON body, so an `extra_body` key would be sent to - the API verbatim and a genuine vendor kwarg would arrive nested one level - too deep. + For a provider listed in `openai_compatible_providers`, `get_optional_params` + nests them under `extra_body`. `BaseLLMHTTPHandler` later merges that back + into the request body, so the wire body does not distinguish the two cases; + this params-level check is what fails if ClinePass drifts back into the list. """ params = litellm.utils.get_optional_params( model="cline-pass/deepseek-v4-flash", From 304d7a2e7715cedb1a8e47b024bf47606e344942 Mon Sep 17 00:00:00 2001 From: Daniel JB Clark Date: Sat, 3 Oct 2026 15:45:57 -0400 Subject: [PATCH 17/20] fix(clinepass): justify the dict-typed overrides for the type-discipline budget The mutable-collection annotations in the ClinePass chat config are all dictated by the OpenAIGPTConfig signatures they override (and the request body handed to its transform_request). Mark each with the repo's own '# mutable-ok: matches the dict-typed base-class signature' so the change adds none to the LIT001 budget. Co-Authored-By: Claude Sonnet 5.5 --- litellm/llms/clinepass/chat/transformation.py | 41 +++++++++++-------- 1 file changed, 24 insertions(+), 17 deletions(-) diff --git a/litellm/llms/clinepass/chat/transformation.py b/litellm/llms/clinepass/chat/transformation.py index f4dc9d13bd3..301ac936f64 100644 --- a/litellm/llms/clinepass/chat/transformation.py +++ b/litellm/llms/clinepass/chat/transformation.py @@ -101,7 +101,7 @@ def _unwrap_response_envelope(raw_response: httpx.Response) -> httpx.Response: ) -def _apply_model_prefix(data: dict) -> dict: +def _apply_model_prefix(data: dict) -> dict: # mutable-ok: request body handed to the dict-typed base transform_request """Restore the ``modelType/model`` qualifier on the outbound model id. Only prefix ids that lost their qualifier, so a cross-provider id @@ -133,8 +133,8 @@ class ClinePassConfig(OpenAIGPTConfig): api_base: str | None, api_key: str | None, model: str, - optional_params: dict, - litellm_params: dict, + optional_params: dict, # mutable-ok: matches the dict-typed base-class signature + litellm_params: dict, # mutable-ok: matches the dict-typed base-class signature stream: bool | None = None, ) -> str: if not api_base: @@ -146,7 +146,9 @@ class ClinePassConfig(OpenAIGPTConfig): return f"{api_base}/chat/completions" - def get_models(self, api_key: str | None = None, api_base: str | None = None) -> list[str]: + def get_models( + self, api_key: str | None = None, api_base: str | None = None + ) -> list[str]: # mutable-ok: matches the dict-typed base-class signature """ClinePass exposes no model catalog. ``GET https://api.cline.bot/api/v1/models`` returns HTTP 404, and the @@ -159,11 +161,11 @@ class ClinePassConfig(OpenAIGPTConfig): def map_openai_params( self, - non_default_params: dict, - optional_params: dict, + non_default_params: dict, # mutable-ok: matches the dict-typed base-class signature + optional_params: dict, # mutable-ok: matches the dict-typed base-class signature model: str, drop_params: bool, - ) -> dict: + ) -> dict: # mutable-ok: matches the dict-typed base-class signature """ClinePass takes the legacy ``max_tokens`` spelling only.""" mapped_params = super().map_openai_params( non_default_params=non_default_params, @@ -178,11 +180,11 @@ class ClinePassConfig(OpenAIGPTConfig): def transform_request( self, model: str, - messages: list[AllMessageValues], - optional_params: dict, - litellm_params: dict, - headers: dict, - ) -> dict: + messages: list[AllMessageValues], # mutable-ok: matches the dict-typed base-class signature + optional_params: dict, # mutable-ok: matches the dict-typed base-class signature + litellm_params: dict, # mutable-ok: matches the dict-typed base-class signature + headers: dict, # mutable-ok: matches the dict-typed base-class signature + ) -> dict: # mutable-ok: matches the dict-typed base-class signature # BaseLLMHTTPHandler builds the body with this synchronous method on # both the sync and the async path, so there is deliberately no # async_transform_request() override -- it would never be called. @@ -201,10 +203,10 @@ class ClinePassConfig(OpenAIGPTConfig): raw_response: httpx.Response, model_response: ModelResponse, logging_obj: LiteLLMLoggingObj, - request_data: dict, - messages: list[AllMessageValues], - optional_params: dict, - litellm_params: dict, + request_data: dict, # mutable-ok: matches the dict-typed base-class signature + messages: list[AllMessageValues], # mutable-ok: matches the dict-typed base-class signature + optional_params: dict, # mutable-ok: matches the dict-typed base-class signature + litellm_params: dict, # mutable-ok: matches the dict-typed base-class signature encoding: "Tokenizer | None", api_key: str | None = None, json_mode: bool | None = None, @@ -228,7 +230,12 @@ class ClinePassConfig(OpenAIGPTConfig): json_mode=json_mode, ) - def get_error_class(self, error_message: str, status_code: int, headers: dict | httpx.Headers) -> BaseLLMException: + def get_error_class( + self, + error_message: str, + status_code: int, + headers: dict | httpx.Headers, # mutable-ok: matches the dict-typed base-class signature + ) -> BaseLLMException: return ClinePassException( message=error_message, status_code=status_code, From ace7be673df91323206dd70d3f9224527ec6ce60 Mon Sep 17 00:00:00 2001 From: Daniel JB Clark Date: Sun, 4 Oct 2026 05:01:28 -0400 Subject: [PATCH 18/20] fix(clinepass): reject realtime credential fallback and isolate provider policy Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- litellm/llms/clinepass/chat/transformation.py | 15 +++- litellm/main.py | 13 +-- litellm/utils.py | 4 +- .../test_clinepass_chat_transformation.py | 86 ++++++++++++++++++- .../test_clinepass_endpoint_guard.py | 40 ++++++++- 5 files changed, 143 insertions(+), 15 deletions(-) diff --git a/litellm/llms/clinepass/chat/transformation.py b/litellm/llms/clinepass/chat/transformation.py index 301ac936f64..6db0819a51a 100644 --- a/litellm/llms/clinepass/chat/transformation.py +++ b/litellm/llms/clinepass/chat/transformation.py @@ -12,10 +12,13 @@ ClinePass is OpenAI-compatible apart from two quirks, both handled here: prefix before the request is built, so a qualifier has to be restored. Documentation: https://docs.cline.bot/ + +Credentials come only from the request's api_key or CLINEPASS_API_KEY. +Realtime endpoints are unsupported and rejected before HTTP dispatch. """ import json -from typing import TYPE_CHECKING, Any, Final +from typing import TYPE_CHECKING, Any, Final, NoReturn import httpx @@ -121,6 +124,16 @@ class ClinePassConfig(OpenAIGPTConfig): module docstring for the two quirks. """ + @staticmethod + def get_realtime_http_config(model: str) -> NoReturn: + from litellm.exceptions import BadRequestError + + raise BadRequestError( + message="ClinePass does not support realtime endpoints", + model=model, + llm_provider="clinepass", + ) + def _get_openai_compatible_provider_info( self, api_base: str | None, api_key: str | None ) -> tuple[str | None, str | None]: diff --git a/litellm/main.py b/litellm/main.py index e8eb20f2060..83cf6b33d6b 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -2448,10 +2448,10 @@ def _complete_aiohttp_openai( ) -def _complete_clinepass(ctx: _CompletionDispatchContext) -> _CompletionDispatchResult: +def _complete_http_provider(ctx: _CompletionDispatchContext) -> _CompletionDispatchResult: acompletion: Final = ctx.acompletion - api_base = ctx.api_base - api_key = ctx.api_key + api_base: Final = ctx.api_base + api_key: Final = ctx.api_key client: Final = _dispatch_client_http(ctx) custom_llm_provider: Final = ctx.custom_llm_provider headers: Final = ctx.headers @@ -2466,11 +2466,6 @@ def _complete_clinepass(ctx: _CompletionDispatchContext) -> _CompletionDispatchR stream: Final = ctx.stream timeout: Final = ctx.timeout - api_key = api_key or get_secret_str("CLINEPASS_API_KEY") - - api_base = api_base or litellm.api_base or get_secret_str("CLINEPASS_API_BASE") or "https://api.cline.bot/api/v1" - - ## COMPLETION CALL response: Final = base_llm_http_handler.completion( model=model, messages=messages, # pyright: ignore[reportUnknownArgumentType] # ctx.messages is list[Unknown] @@ -5966,7 +5961,7 @@ def completion( elif custom_llm_provider == "cometapi": response = _complete_cometapi(_dispatch_ctx) elif custom_llm_provider == "clinepass": - response = _complete_clinepass(_dispatch_ctx) + response = _complete_http_provider(_dispatch_ctx) elif custom_llm_provider == "minimax": response = _complete_minimax(_dispatch_ctx) elif custom_llm_provider == "hosted_vllm": diff --git a/litellm/utils.py b/litellm/utils.py index ab840f4016b..7398a3415ad 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -8537,7 +8537,7 @@ class ProviderConfigManager: LlmProviders.EDENAI: (litellm.EdenAIChatConfig, False), LlmProviders.FAL_AI: (litellm.FalAIChatConfig, False), LlmProviders.COMETAPI: (lambda: litellm.CometAPIConfig(), False), - LlmProviders.CLINEPASS: (lambda: litellm.ClinePassConfig(), False), + LlmProviders.CLINEPASS: (litellm.ClinePassConfig, False), LlmProviders.DATAROBOT: (lambda: litellm.DataRobotConfig(), False), LlmProviders.GEMINI: (lambda: litellm.GoogleAIStudioGeminiConfig(), False), LlmProviders.AI21: (lambda: litellm.AI21ChatConfig(), False), @@ -9744,6 +9744,8 @@ class ProviderConfigManager: (POST /realtime/client_secrets and POST /realtime/calls). """ + if LlmProviders.CLINEPASS == provider: + return litellm.ClinePassConfig.get_realtime_http_config(model=model) if LlmProviders.OPENAI == provider: from litellm.llms.openai.realtime.http_transformation import ( OpenAIRealtimeHTTPConfig, diff --git a/tests/unit/llms/clinepass/chat/test_clinepass_chat_transformation.py b/tests/unit/llms/clinepass/chat/test_clinepass_chat_transformation.py index 2b11898f0bc..e1fc3a2bf3f 100644 --- a/tests/unit/llms/clinepass/chat/test_clinepass_chat_transformation.py +++ b/tests/unit/llms/clinepass/chat/test_clinepass_chat_transformation.py @@ -8,6 +8,7 @@ shipped as dead code. """ import json +from typing import Final from unittest.mock import patch import httpx @@ -58,6 +59,69 @@ def _clinepass_env(monkeypatch): monkeypatch.delenv("CLINEPASS_API_BASE", raising=False) +@pytest.fixture +def unrelated_credentials(monkeypatch): + for env_name in ("OPENAI_API_KEY", "ANTHROPIC_API_KEY", "GROQ_API_KEY", "OPENROUTER_API_KEY"): + monkeypatch.setenv(env_name, f"sk-unrelated-{env_name}") + for attribute in ("api_key", "openai_key", "anthropic_key"): + monkeypatch.setattr(litellm, attribute, f"sk-unrelated-{attribute}") + + +@pytest.mark.parametrize("credential_source", ["missing", "environment", "explicit"]) +@pytest.mark.parametrize("custom_llm_provider", [None, "clinepass"]) +def test_chat_sends_only_clinepass_credentials( + monkeypatch, unrelated_credentials, credential_source, custom_llm_provider +): + if credential_source == "missing": + monkeypatch.delenv("CLINEPASS_API_KEY", raising=False) + explicit_key: Final = "cp-request-key" if credential_source == "explicit" else None + captured = {} + + def fake_post(self, url, *args, **kwargs): + captured["url"] = str(url) + captured["headers"] = httpx.Headers(kwargs["headers"]) + return _response(ENVELOPED_COMPLETION) + + with patch.object(HTTPHandler, "post", fake_post): + response = litellm.completion( + model="deepseek-v4-flash" if custom_llm_provider else "clinepass/deepseek-v4-flash", + custom_llm_provider=custom_llm_provider, + api_key=explicit_key, + messages=[{"role": "user", "content": "ping"}], + ) + + expected_key: Final = explicit_key or (API_KEY if credential_source == "environment" else None) + assert captured["url"] == "https://api.cline.bot/api/v1/chat/completions" + assert captured["headers"].get("authorization") == (f"Bearer {expected_key}" if expected_key else None) + assert response.choices[0].message.content == "pong" + + +@pytest.mark.parametrize("credential_source", ["missing", "environment", "explicit"]) +@pytest.mark.asyncio +async def test_async_chat_sends_only_clinepass_credentials(monkeypatch, unrelated_credentials, credential_source): + if credential_source == "missing": + monkeypatch.delenv("CLINEPASS_API_KEY", raising=False) + explicit_key: Final = "cp-request-key" if credential_source == "explicit" else None + captured = {} + + async def fake_post(self, url, *args, **kwargs): + captured["url"] = str(url) + captured["headers"] = httpx.Headers(kwargs["headers"]) + return _response(ENVELOPED_COMPLETION) + + with patch.object(AsyncHTTPHandler, "post", fake_post): + response = await litellm.acompletion( + model="clinepass/deepseek-v4-flash", + api_key=explicit_key, + messages=[{"role": "user", "content": "ping"}], + ) + + expected_key: Final = explicit_key or (API_KEY if credential_source == "environment" else None) + assert captured["url"] == "https://api.cline.bot/api/v1/chat/completions" + assert captured["headers"].get("authorization") == (f"Bearer {expected_key}" if expected_key else None) + assert response.choices[0].message.content == "pong" + + # -------------------------------------------------------------------------- # Registration / routing # -------------------------------------------------------------------------- @@ -308,8 +372,13 @@ async def test_acompletion_unwraps_envelope_and_prefixes_model(): assert response.choices[0].message.content == "pong" -def test_completion_streaming_is_not_unwrapped(): +@pytest.mark.parametrize("credential_source", ["missing", "environment", "explicit"]) +def test_completion_streaming_is_not_unwrapped(monkeypatch, unrelated_credentials, credential_source): """ClinePass does NOT wrap SSE chunks -- they are already OpenAI-shaped.""" + if credential_source == "missing": + monkeypatch.delenv("CLINEPASS_API_KEY", raising=False) + explicit_key: Final = "cp-stream-key" if credential_source == "explicit" else None + captured = {} chunks = [ { "id": "chatcmpl-test", @@ -332,6 +401,7 @@ def test_completion_streaming_is_not_unwrapped(): body = "".join(f"data: {json.dumps(c)}\n\n" for c in chunks) + "data: [DONE]\n\n" def fake_post(self, url, *args, **kwargs): + captured["authorization"] = httpx.Headers(kwargs["headers"]).get("authorization") return httpx.Response( 200, content=body.encode(), @@ -343,6 +413,7 @@ def test_completion_streaming_is_not_unwrapped(): stream = litellm.completion( model="clinepass/deepseek-v4-flash", messages=[{"role": "user", "content": "count"}], + api_key=explicit_key, max_tokens=4000, stream=True, ) @@ -357,10 +428,17 @@ def test_completion_streaming_is_not_unwrapped(): assert text == "one two three" assert finish_reason == "stop" + expected_key: Final = explicit_key or (API_KEY if credential_source == "environment" else None) + assert captured["authorization"] == (f"Bearer {expected_key}" if expected_key else None) +@pytest.mark.parametrize("credential_source", ["missing", "environment", "explicit"]) @pytest.mark.asyncio -async def test_acompletion_streaming_is_not_unwrapped(): +async def test_acompletion_streaming_is_not_unwrapped(monkeypatch, unrelated_credentials, credential_source): + if credential_source == "missing": + monkeypatch.delenv("CLINEPASS_API_KEY", raising=False) + explicit_key: Final = "cp-stream-key" if credential_source == "explicit" else None + captured = {} chunks = [ { "id": "chatcmpl-test", @@ -383,6 +461,7 @@ async def test_acompletion_streaming_is_not_unwrapped(): body = "".join(f"data: {json.dumps(c)}\n\n" for c in chunks) + "data: [DONE]\n\n" async def fake_post(self, url, *args, **kwargs): + captured["authorization"] = httpx.Headers(kwargs["headers"]).get("authorization") return httpx.Response( 200, content=body.encode(), @@ -394,6 +473,7 @@ async def test_acompletion_streaming_is_not_unwrapped(): stream = await litellm.acompletion( model="clinepass/deepseek-v4-flash", messages=[{"role": "user", "content": "count"}], + api_key=explicit_key, max_tokens=4000, stream=True, ) @@ -408,6 +488,8 @@ async def test_acompletion_streaming_is_not_unwrapped(): assert text == "one two three" assert finish_reason == "stop" + expected_key: Final = explicit_key or (API_KEY if credential_source == "environment" else None) + assert captured["authorization"] == (f"Bearer {expected_key}" if expected_key else None) def test_completion_streaming_tool_call_reassembly(): diff --git a/tests/unit/llms/clinepass/test_clinepass_endpoint_guard.py b/tests/unit/llms/clinepass/test_clinepass_endpoint_guard.py index a961641ce86..35c490db314 100644 --- a/tests/unit/llms/clinepass/test_clinepass_endpoint_guard.py +++ b/tests/unit/llms/clinepass/test_clinepass_endpoint_guard.py @@ -13,6 +13,7 @@ leaves the process at all, and that the OpenAI credential is never transmitted. """ import contextlib +from typing import Final import httpx import pytest @@ -36,10 +37,17 @@ def no_request_allowed(monkeypatch): attempted.append((str(request.url), request.headers.get("authorization", ""))) raise AssertionError(f"outbound request attempted to {request.url}") + def record_handler_and_block(self, url, *args, **kwargs): + attempted.append((str(url), httpx.Headers(kwargs.get("headers") or {}).get("authorization", ""))) + raise AssertionError(f"outbound request attempted to {url}") + + async def record_async_handler_and_block(self, url, *args, **kwargs): + record_handler_and_block(self, url, *args, **kwargs) + monkeypatch.setattr(httpx.Client, "send", record_and_block, raising=True) monkeypatch.setattr(httpx.AsyncClient, "send", record_and_block, raising=True) - for handler in (HTTPHandler, AsyncHTTPHandler): - monkeypatch.setattr(handler, "post", record_and_block, raising=True) + monkeypatch.setattr(HTTPHandler, "post", record_handler_and_block, raising=True) + monkeypatch.setattr(AsyncHTTPHandler, "post", record_async_handler_and_block, raising=True) monkeypatch.setenv("OPENAI_API_KEY", SENTINEL_OPENAI_KEY) monkeypatch.setenv("CLINEPASS_API_KEY", "cp-test-key") @@ -152,3 +160,31 @@ def test_chat_does_not_fall_back_to_the_global_litellm_api_key(monkeypatch): assert sent, "the chat request never reached the transport, so nothing was checked" assert all(SENTINEL_OPENAI_KEY not in authorization for _, authorization in sent) + + +@pytest.mark.parametrize("clinepass_key", [None, "cp-test-key"]) +@pytest.mark.parametrize("explicit_key", [None, "cp-explicit-key"]) +@pytest.mark.parametrize("endpoint", ["client_secret", "transcription_session", "calls"]) +@pytest.mark.asyncio +async def test_realtime_rejects_clinepass_before_credential_fallback( + monkeypatch, no_request_allowed, clinepass_key, explicit_key, endpoint +): + if clinepass_key is None: + monkeypatch.delenv("CLINEPASS_API_KEY", raising=False) + else: + monkeypatch.setenv("CLINEPASS_API_KEY", clinepass_key) + monkeypatch.setattr(litellm, "api_key", SENTINEL_OPENAI_KEY) + monkeypatch.setattr(litellm, "openai_key", SENTINEL_OPENAI_KEY) + kwargs: Final = {"model": "clinepass/deepseek-v4-flash", "api_key": explicit_key} + request: Final = ( + litellm.acreate_realtime_client_secret(**kwargs) + if endpoint == "client_secret" + else litellm.acreate_realtime_transcription_session(**kwargs) + if endpoint == "transcription_session" + else litellm.arealtime_calls(openai_ephemeral_key=SENTINEL_OPENAI_KEY, sdp_body=b"v=0\r\n", **kwargs) + ) + + with pytest.raises(litellm.BadRequestError, match="ClinePass does not support realtime endpoints"): + await request + + assert no_request_allowed == [] From 497dc8a7aa584a3864325f423583f9b123db3414 Mon Sep 17 00:00:00 2001 From: Daniel JB Clark Date: Sun, 4 Oct 2026 05:44:38 -0400 Subject: [PATCH 19/20] fix(clinepass): isolate moderation and responses websocket credentials Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- litellm/llms/clinepass/chat/transformation.py | 14 ++- litellm/main.py | 2 + litellm/responses/main.py | 9 +- litellm/responses/streaming_iterator.py | 7 +- .../test_clinepass_chat_transformation.py | 86 +++++++++++++++++++ .../test_clinepass_endpoint_guard.py | 69 +++++++++++++++ 6 files changed, 180 insertions(+), 7 deletions(-) diff --git a/litellm/llms/clinepass/chat/transformation.py b/litellm/llms/clinepass/chat/transformation.py index 6db0819a51a..c023335d876 100644 --- a/litellm/llms/clinepass/chat/transformation.py +++ b/litellm/llms/clinepass/chat/transformation.py @@ -14,7 +14,7 @@ ClinePass is OpenAI-compatible apart from two quirks, both handled here: Documentation: https://docs.cline.bot/ Credentials come only from the request's api_key or CLINEPASS_API_KEY. -Realtime endpoints are unsupported and rejected before HTTP dispatch. +Moderation and realtime endpoints are unsupported and rejected before dispatch. """ import json @@ -134,6 +134,18 @@ class ClinePassConfig(OpenAIGPTConfig): llm_provider="clinepass", ) + @staticmethod + def validate_moderation(model: str | None, custom_llm_provider: str | None = None) -> None: + if custom_llm_provider != "clinepass" and not (model or "").startswith("clinepass/"): + return + from litellm.exceptions import BadRequestError + + raise BadRequestError( + message="ClinePass does not support moderation endpoints", + model=model or "", + llm_provider="clinepass", + ) + def _get_openai_compatible_provider_info( self, api_base: str | None, api_key: str | None ) -> tuple[str | None, str | None]: diff --git a/litellm/main.py b/litellm/main.py index 83cf6b33d6b..aa1a632b0e7 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -7808,6 +7808,7 @@ def adapter_completion(*, adapter_id: str, **kwargs) -> BaseModel | AdapterCompl def moderation(input: str, model: str | None = None, api_key: str | None = None, **kwargs) -> OpenAIModerationResponse: + litellm.ClinePassConfig.validate_moderation(model=model, custom_llm_provider=kwargs.get("custom_llm_provider")) # only supports open ai for now api_key = api_key or litellm.api_key or litellm.openai_key or get_secret_str("OPENAI_API_KEY") @@ -7842,6 +7843,7 @@ async def amoderation( ) -> OpenAIModerationResponse: from openai import AsyncOpenAI + litellm.ClinePassConfig.validate_moderation(model=model, custom_llm_provider=custom_llm_provider) # only supports open ai for now api_key = api_key or litellm.api_key or litellm.openai_key or get_secret_str("OPENAI_API_KEY") optional_params: Final = GenericLiteLLMParams(**kwargs) diff --git a/litellm/responses/main.py b/litellm/responses/main.py index 14c12fc571d..b595ee731d7 100644 --- a/litellm/responses/main.py +++ b/litellm/responses/main.py @@ -2405,10 +2405,13 @@ async def _aresponses_websocket( resolved_api_key: Final = ( dynamic_api_key + or api_key or litellm_params.api_key - or litellm.api_key - or litellm.openai_key - or get_secret_str("OPENAI_API_KEY") + or ( + (litellm.api_key or litellm.openai_key or get_secret_str("OPENAI_API_KEY")) + if responses_api_provider_config is not None + else None + ) ) # Extract params that we're passing explicitly to avoid duplicates in **kwargs diff --git a/litellm/responses/streaming_iterator.py b/litellm/responses/streaming_iterator.py index 5e045c3e84f..475acd5b967 100644 --- a/litellm/responses/streaming_iterator.py +++ b/litellm/responses/streaming_iterator.py @@ -2824,9 +2824,10 @@ class ManagedResponsesWebSocketHandler: def _inject_credentials(self, call_kwargs: dict[str, object], model: str | None = None) -> None: """Inject connection-level credentials and metadata into call_kwargs.""" - if self.api_key is not None: + same_provider: Final = self._same_provider(model) + if self.api_key is not None and same_provider: call_kwargs["api_key"] = self.api_key - if self.api_base is not None: + if self.api_base is not None and same_provider: call_kwargs["api_base"] = self.api_base if self.timeout is not None: call_kwargs["timeout"] = self.timeout @@ -2835,7 +2836,7 @@ class ManagedResponsesWebSocketHandler: # (e.g., connection is vertex_ai but event says openai/gpt-4), let litellm # re-resolve from the model string. Same-provider model variants (e.g., # vertex_ai/gemini-2.0 -> vertex_ai/gemini-1.5) still inherit the provider. - if self.custom_llm_provider is not None and self._same_provider(model): + if self.custom_llm_provider is not None and same_provider: call_kwargs["custom_llm_provider"] = self.custom_llm_provider if self.litellm_metadata: call_kwargs["litellm_metadata"] = dict(self.litellm_metadata) diff --git a/tests/unit/llms/clinepass/chat/test_clinepass_chat_transformation.py b/tests/unit/llms/clinepass/chat/test_clinepass_chat_transformation.py index e1fc3a2bf3f..85115d754a2 100644 --- a/tests/unit/llms/clinepass/chat/test_clinepass_chat_transformation.py +++ b/tests/unit/llms/clinepass/chat/test_clinepass_chat_transformation.py @@ -13,14 +13,17 @@ from unittest.mock import patch import httpx import pytest +from starlette.websockets import WebSocket import litellm +from litellm.litellm_core_utils.litellm_logging import Logging from litellm.llms.clinepass.chat.transformation import ( ClinePassConfig, _apply_model_prefix, _unwrap_response_envelope, ) from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler +from litellm.responses.main import _aresponses_websocket from litellm.types.utils import LlmProviders from litellm.utils import ProviderConfigManager @@ -122,6 +125,89 @@ async def test_async_chat_sends_only_clinepass_credentials(monkeypatch, unrelate assert response.choices[0].message.content == "pong" +@pytest.mark.parametrize("credential_source", ["missing", "environment", "explicit"]) +@pytest.mark.parametrize("connection_provider", ["clinepass", "mistral"]) +@pytest.mark.asyncio +async def test_managed_responses_websocket_sends_only_clinepass_credentials( + monkeypatch, unrelated_credentials, credential_source, connection_provider +): + if credential_source == "missing": + monkeypatch.delenv("CLINEPASS_API_KEY", raising=False) + connection_key: Final = ( + "sk-unrelated-mistral" + if connection_provider == "mistral" + else "cp-request-key" + if credential_source == "explicit" + else None + ) + sent = [] + received = [] + lifecycle = iter(({"type": "websocket.connect"}, {"type": "websocket.disconnect", "code": 1000})) + + async def receive(): + return next(lifecycle) + + async def send(message): + if message["type"] == "websocket.send": + received.append(json.loads(message["text"])) + + websocket: Final = WebSocket( + scope={"type": "websocket", "path": "/v1/responses", "headers": [], "query_string": b""}, + receive=receive, + send=send, + ) + await websocket.accept() + chunk: Final = { + "id": "chatcmpl-test", + "object": "chat.completion.chunk", + "created": 1, + "model": "cline-pass/deepseek-v4-flash", + "choices": [{"index": 0, "delta": {"role": "assistant", "content": "pong"}, "finish_reason": "stop"}], + } + + async def fake_post(self, url, *args, **kwargs): + sent.append((str(url), httpx.Headers(kwargs["headers"]).get("authorization"))) + return httpx.Response( + 200, + content=f"data: {json.dumps(chunk)}\n\ndata: [DONE]\n\n".encode(), + headers={"content-type": "text/event-stream"}, + request=httpx.Request("POST", str(url)), + ) + + with patch.object(AsyncHTTPHandler, "post", fake_post): + result = await _aresponses_websocket.__wrapped__( + model=f"{connection_provider}/deepseek-v4-flash", + websocket=websocket, + api_key=connection_key, + first_message=json.dumps( + {"type": "response.create", "model": "clinepass/deepseek-v4-flash", "input": "ping"} + ), + litellm_logging_obj=Logging( + model=f"{connection_provider}/deepseek-v4-flash", + messages=[], + stream=True, + call_type="aresponses", + start_time=0, + litellm_call_id="cp-ws-test", + function_id="cp-ws-test", + ), + ) + + expected_key: Final = ( + connection_key + if connection_provider == "clinepass" and credential_source == "explicit" + else API_KEY + if credential_source != "missing" + else None + ) + assert sent == [ + ("https://api.cline.bot/api/v1/chat/completions", f"Bearer {expected_key}" if expected_key else None) + ] + assert result is None + assert "response.completed" in [event["type"] for event in received] + assert "error" not in [event["type"] for event in received] + + # -------------------------------------------------------------------------- # Registration / routing # -------------------------------------------------------------------------- diff --git a/tests/unit/llms/clinepass/test_clinepass_endpoint_guard.py b/tests/unit/llms/clinepass/test_clinepass_endpoint_guard.py index 35c490db314..0a06a15cc1a 100644 --- a/tests/unit/llms/clinepass/test_clinepass_endpoint_guard.py +++ b/tests/unit/llms/clinepass/test_clinepass_endpoint_guard.py @@ -17,6 +17,7 @@ from typing import Final import httpx import pytest +from openai import AsyncOpenAI, OpenAI import litellm from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler @@ -69,6 +70,74 @@ def test_speech_makes_no_outbound_request(no_request_allowed): assert no_request_allowed == [] +@pytest.mark.parametrize("clinepass_key", [None, "cp-test-key"]) +@pytest.mark.parametrize("custom_llm_provider", [None, "clinepass"]) +@pytest.mark.parametrize("explicit_key", [None, "cp-explicit-key"]) +def test_moderation_rejects_clinepass_before_credential_fallback( + monkeypatch, no_request_allowed, clinepass_key, custom_llm_provider, explicit_key +): + if clinepass_key is None: + monkeypatch.delenv("CLINEPASS_API_KEY", raising=False) + monkeypatch.setattr(litellm, "api_key", SENTINEL_OPENAI_KEY) + model: Final = "deepseek-v4-flash" if custom_llm_provider else "clinepass/deepseek-v4-flash" + + with pytest.raises(litellm.BadRequestError, match="ClinePass does not support moderation endpoints"): + litellm.moderation(model=model, input="hi", custom_llm_provider=custom_llm_provider, api_key=explicit_key) + + assert no_request_allowed == [] + + +@pytest.mark.parametrize("clinepass_key", [None, "cp-test-key"]) +@pytest.mark.parametrize("custom_llm_provider", [None, "clinepass"]) +@pytest.mark.parametrize("explicit_key", [None, "cp-explicit-key"]) +@pytest.mark.asyncio +async def test_async_moderation_rejects_clinepass_before_credential_fallback( + monkeypatch, no_request_allowed, clinepass_key, custom_llm_provider, explicit_key +): + if clinepass_key is None: + monkeypatch.delenv("CLINEPASS_API_KEY", raising=False) + monkeypatch.setattr(litellm, "api_key", SENTINEL_OPENAI_KEY) + model: Final = "deepseek-v4-flash" if custom_llm_provider else "clinepass/deepseek-v4-flash" + + with pytest.raises(litellm.BadRequestError, match="ClinePass does not support moderation endpoints"): + await litellm.amoderation( + model=model, input="hi", custom_llm_provider=custom_llm_provider, api_key=explicit_key + ) + + assert no_request_allowed == [] + + +@pytest.mark.parametrize("async_mode", [False, True]) +@pytest.mark.asyncio +async def test_openai_moderation_remains_supported(async_mode): + sent = [] + + def respond(request): + sent.append((str(request.url), request.headers["authorization"])) + return httpx.Response( + 200, + json={ + "id": "modr-test", + "model": "moderation-test", + "results": [{"flagged": False, "categories": {"violence": False}, "category_scores": {"violence": 0}}], + }, + ) + + transport: Final = httpx.MockTransport(respond) + if async_mode: + async with AsyncOpenAI( + api_key="sk-moderation-test", http_client=httpx.AsyncClient(transport=transport) + ) as client: + response = await litellm.amoderation(model="openai/moderation-test", input="hi", client=client) + else: + with OpenAI(api_key="sk-moderation-test", http_client=httpx.Client(transport=transport)) as client: + response = litellm.moderation(model="moderation-test", input="hi", client=client) + + assert sent == [("https://api.openai.com/v1/moderations", "Bearer sk-moderation-test")] + assert response.id == "modr-test" + assert response.results[0].flagged is False + + def test_transcription_makes_no_outbound_request(no_request_allowed, tmp_path): audio = tmp_path / "a.mp3" audio.write_bytes(b"\x00\x00") From 0bc31c076f2faed3e8de64543b53f1393e2e649e Mon Sep 17 00:00:00 2001 From: Daniel JB Clark Date: Sun, 4 Oct 2026 06:08:47 -0400 Subject: [PATCH 20/20] fix(clinepass): pin authenticated websocket models and type moderation routing Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- litellm/main.py | 5 +- litellm/responses/streaming_iterator.py | 15 +++-- .../test_clinepass_chat_transformation.py | 57 ++++++++++++++++--- 3 files changed, 65 insertions(+), 12 deletions(-) diff --git a/litellm/main.py b/litellm/main.py index aa1a632b0e7..49ee53e1c38 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -7808,7 +7808,10 @@ def adapter_completion(*, adapter_id: str, **kwargs) -> BaseModel | AdapterCompl def moderation(input: str, model: str | None = None, api_key: str | None = None, **kwargs) -> OpenAIModerationResponse: - litellm.ClinePassConfig.validate_moderation(model=model, custom_llm_provider=kwargs.get("custom_llm_provider")) + custom_llm_provider: Final[object] = kwargs.get("custom_llm_provider") + litellm.ClinePassConfig.validate_moderation( + model=model, custom_llm_provider=custom_llm_provider if isinstance(custom_llm_provider, str) else None + ) # only supports open ai for now api_key = api_key or litellm.api_key or litellm.openai_key or get_secret_str("OPENAI_API_KEY") diff --git a/litellm/responses/streaming_iterator.py b/litellm/responses/streaming_iterator.py index 475acd5b967..02e43bc8bb8 100644 --- a/litellm/responses/streaming_iterator.py +++ b/litellm/responses/streaming_iterator.py @@ -2961,11 +2961,18 @@ class ManagedResponsesWebSocketHandler: call_kwargs: Final = self._build_base_call_kwargs(msg_obj) call_kwargs["stream"] = True - # A frame that repeats the connection's public alias (model_group) must - # reuse the router-resolved self.model; passing the alias raw to - # litellm.aresponses fails in get_llm_provider. A genuinely different - # provider-prefixed per-frame model is still honored. requested_model: Final[str | None] = _optional_str(call_kwargs.pop("model", None)) + authorized_models: Final = (self.model, self.model_group, f"{self.custom_llm_provider}/{self.model}") + if ( + self.user_api_key_dict is not None + and requested_model is not None + and requested_model not in authorized_models + ): + await self._send_error( + "Changing models requires a new authorized WebSocket connection", + error_type="invalid_request_error", + ) + return model: Final[str] = ( self.model if requested_model is None or requested_model == self.model_group else requested_model ) diff --git a/tests/unit/llms/clinepass/chat/test_clinepass_chat_transformation.py b/tests/unit/llms/clinepass/chat/test_clinepass_chat_transformation.py index 85115d754a2..2bb24e6e541 100644 --- a/tests/unit/llms/clinepass/chat/test_clinepass_chat_transformation.py +++ b/tests/unit/llms/clinepass/chat/test_clinepass_chat_transformation.py @@ -23,6 +23,7 @@ from litellm.llms.clinepass.chat.transformation import ( _unwrap_response_envelope, ) from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler +from litellm.proxy._types import UserAPIKeyAuth from litellm.responses.main import _aresponses_websocket from litellm.types.utils import LlmProviders from litellm.utils import ProviderConfigManager @@ -127,9 +128,10 @@ async def test_async_chat_sends_only_clinepass_credentials(monkeypatch, unrelate @pytest.mark.parametrize("credential_source", ["missing", "environment", "explicit"]) @pytest.mark.parametrize("connection_provider", ["clinepass", "mistral"]) +@pytest.mark.parametrize("changed_model", [None, "openai/gpt-4o", "clinepass/unauthorized-model"]) @pytest.mark.asyncio async def test_managed_responses_websocket_sends_only_clinepass_credentials( - monkeypatch, unrelated_credentials, credential_source, connection_provider + monkeypatch, unrelated_credentials, credential_source, connection_provider, changed_model ): if credential_source == "missing": monkeypatch.delenv("CLINEPASS_API_KEY", raising=False) @@ -142,7 +144,29 @@ async def test_managed_responses_websocket_sends_only_clinepass_credentials( ) sent = [] received = [] - lifecycle = iter(({"type": "websocket.connect"}, {"type": "websocket.disconnect", "code": 1000})) + foreign_requests = [] + lifecycle = iter( + ( + {"type": "websocket.connect"}, + *( + ( + { + "type": "websocket.receive", + "text": json.dumps({"type": "response.create", "model": changed_model, "input": "hi"}), + }, + ) + if changed_model is not None + else () + ), + {"type": "websocket.disconnect", "code": 1000}, + ) + ) + + async def block_foreign_request(self, request, *args, **kwargs): + foreign_requests.append(str(request.url)) + raise AssertionError("Unexpected provider HTTP request") + + monkeypatch.setattr(httpx.AsyncClient, "send", block_foreign_request) async def receive(): return next(lifecycle) @@ -179,6 +203,11 @@ async def test_managed_responses_websocket_sends_only_clinepass_credentials( model=f"{connection_provider}/deepseek-v4-flash", websocket=websocket, api_key=connection_key, + user_api_key_dict=( + UserAPIKeyAuth(models=[f"{connection_provider}/deepseek-v4-flash"]) + if changed_model is not None + else None + ), first_message=json.dumps( {"type": "response.create", "model": "clinepass/deepseek-v4-flash", "input": "ping"} ), @@ -200,12 +229,26 @@ async def test_managed_responses_websocket_sends_only_clinepass_credentials( if credential_source != "missing" else None ) - assert sent == [ - ("https://api.cline.bot/api/v1/chat/completions", f"Bearer {expected_key}" if expected_key else None) - ] + assert sent == ( + [] + if changed_model is not None and connection_provider == "mistral" + else [("https://api.cline.bot/api/v1/chat/completions", f"Bearer {expected_key}" if expected_key else None)] + ) assert result is None - assert "response.completed" in [event["type"] for event in received] - assert "error" not in [event["type"] for event in received] + errors: Final = [event["error"] for event in received if event["type"] == "error"] + assert errors == ( + [ + { + "type": "invalid_request_error", + "message": "Changing models requires a new authorized WebSocket connection", + } + ] + * (2 if connection_provider == "mistral" else 1) + if changed_model is not None + else [] + ) + assert foreign_requests == [] + assert ("response.completed" in [event["type"] for event in received]) == bool(sent) # --------------------------------------------------------------------------