fix(clinepass): isolate moderation and responses websocket credentials

Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com>
This commit is contained in:
Daniel JB Clark 2026-10-04 05:44:38 -04:00
parent ace7be673d
commit 497dc8a7aa
No known key found for this signature in database
6 changed files with 180 additions and 7 deletions

View file

@ -14,7 +14,7 @@ ClinePass is OpenAI-compatible apart from two quirks, both handled here:
Documentation: https://docs.cline.bot/
Credentials come only from the request's api_key or CLINEPASS_API_KEY.
Realtime endpoints are unsupported and rejected before HTTP dispatch.
Moderation and realtime endpoints are unsupported and rejected before dispatch.
"""
import json
@ -134,6 +134,18 @@ class ClinePassConfig(OpenAIGPTConfig):
llm_provider="clinepass",
)
@staticmethod
def validate_moderation(model: str | None, custom_llm_provider: str | None = None) -> None:
if custom_llm_provider != "clinepass" and not (model or "").startswith("clinepass/"):
return
from litellm.exceptions import BadRequestError
raise BadRequestError(
message="ClinePass does not support moderation endpoints",
model=model or "",
llm_provider="clinepass",
)
def _get_openai_compatible_provider_info(
self, api_base: str | None, api_key: str | None
) -> tuple[str | None, str | None]:

View file

@ -7808,6 +7808,7 @@ def adapter_completion(*, adapter_id: str, **kwargs) -> BaseModel | AdapterCompl
def moderation(input: str, model: str | None = None, api_key: str | None = None, **kwargs) -> OpenAIModerationResponse:
litellm.ClinePassConfig.validate_moderation(model=model, custom_llm_provider=kwargs.get("custom_llm_provider"))
# only supports open ai for now
api_key = api_key or litellm.api_key or litellm.openai_key or get_secret_str("OPENAI_API_KEY")
@ -7842,6 +7843,7 @@ async def amoderation(
) -> OpenAIModerationResponse:
from openai import AsyncOpenAI
litellm.ClinePassConfig.validate_moderation(model=model, custom_llm_provider=custom_llm_provider)
# only supports open ai for now
api_key = api_key or litellm.api_key or litellm.openai_key or get_secret_str("OPENAI_API_KEY")
optional_params: Final = GenericLiteLLMParams(**kwargs)

View file

@ -2405,10 +2405,13 @@ async def _aresponses_websocket(
resolved_api_key: Final = (
dynamic_api_key
or api_key
or litellm_params.api_key
or litellm.api_key
or litellm.openai_key
or get_secret_str("OPENAI_API_KEY")
or (
(litellm.api_key or litellm.openai_key or get_secret_str("OPENAI_API_KEY"))
if responses_api_provider_config is not None
else None
)
)
# Extract params that we're passing explicitly to avoid duplicates in **kwargs

View file

@ -2824,9 +2824,10 @@ class ManagedResponsesWebSocketHandler:
def _inject_credentials(self, call_kwargs: dict[str, object], model: str | None = None) -> None:
"""Inject connection-level credentials and metadata into call_kwargs."""
if self.api_key is not None:
same_provider: Final = self._same_provider(model)
if self.api_key is not None and same_provider:
call_kwargs["api_key"] = self.api_key
if self.api_base is not None:
if self.api_base is not None and same_provider:
call_kwargs["api_base"] = self.api_base
if self.timeout is not None:
call_kwargs["timeout"] = self.timeout
@ -2835,7 +2836,7 @@ class ManagedResponsesWebSocketHandler:
# (e.g., connection is vertex_ai but event says openai/gpt-4), let litellm
# re-resolve from the model string. Same-provider model variants (e.g.,
# vertex_ai/gemini-2.0 -> vertex_ai/gemini-1.5) still inherit the provider.
if self.custom_llm_provider is not None and self._same_provider(model):
if self.custom_llm_provider is not None and same_provider:
call_kwargs["custom_llm_provider"] = self.custom_llm_provider
if self.litellm_metadata:
call_kwargs["litellm_metadata"] = dict(self.litellm_metadata)

View file

@ -13,14 +13,17 @@ from unittest.mock import patch
import httpx
import pytest
from starlette.websockets import WebSocket
import litellm
from litellm.litellm_core_utils.litellm_logging import Logging
from litellm.llms.clinepass.chat.transformation import (
ClinePassConfig,
_apply_model_prefix,
_unwrap_response_envelope,
)
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler
from litellm.responses.main import _aresponses_websocket
from litellm.types.utils import LlmProviders
from litellm.utils import ProviderConfigManager
@ -122,6 +125,89 @@ async def test_async_chat_sends_only_clinepass_credentials(monkeypatch, unrelate
assert response.choices[0].message.content == "pong"
@pytest.mark.parametrize("credential_source", ["missing", "environment", "explicit"])
@pytest.mark.parametrize("connection_provider", ["clinepass", "mistral"])
@pytest.mark.asyncio
async def test_managed_responses_websocket_sends_only_clinepass_credentials(
monkeypatch, unrelated_credentials, credential_source, connection_provider
):
if credential_source == "missing":
monkeypatch.delenv("CLINEPASS_API_KEY", raising=False)
connection_key: Final = (
"sk-unrelated-mistral"
if connection_provider == "mistral"
else "cp-request-key"
if credential_source == "explicit"
else None
)
sent = []
received = []
lifecycle = iter(({"type": "websocket.connect"}, {"type": "websocket.disconnect", "code": 1000}))
async def receive():
return next(lifecycle)
async def send(message):
if message["type"] == "websocket.send":
received.append(json.loads(message["text"]))
websocket: Final = WebSocket(
scope={"type": "websocket", "path": "/v1/responses", "headers": [], "query_string": b""},
receive=receive,
send=send,
)
await websocket.accept()
chunk: Final = {
"id": "chatcmpl-test",
"object": "chat.completion.chunk",
"created": 1,
"model": "cline-pass/deepseek-v4-flash",
"choices": [{"index": 0, "delta": {"role": "assistant", "content": "pong"}, "finish_reason": "stop"}],
}
async def fake_post(self, url, *args, **kwargs):
sent.append((str(url), httpx.Headers(kwargs["headers"]).get("authorization")))
return httpx.Response(
200,
content=f"data: {json.dumps(chunk)}\n\ndata: [DONE]\n\n".encode(),
headers={"content-type": "text/event-stream"},
request=httpx.Request("POST", str(url)),
)
with patch.object(AsyncHTTPHandler, "post", fake_post):
result = await _aresponses_websocket.__wrapped__(
model=f"{connection_provider}/deepseek-v4-flash",
websocket=websocket,
api_key=connection_key,
first_message=json.dumps(
{"type": "response.create", "model": "clinepass/deepseek-v4-flash", "input": "ping"}
),
litellm_logging_obj=Logging(
model=f"{connection_provider}/deepseek-v4-flash",
messages=[],
stream=True,
call_type="aresponses",
start_time=0,
litellm_call_id="cp-ws-test",
function_id="cp-ws-test",
),
)
expected_key: Final = (
connection_key
if connection_provider == "clinepass" and credential_source == "explicit"
else API_KEY
if credential_source != "missing"
else None
)
assert sent == [
("https://api.cline.bot/api/v1/chat/completions", f"Bearer {expected_key}" if expected_key else None)
]
assert result is None
assert "response.completed" in [event["type"] for event in received]
assert "error" not in [event["type"] for event in received]
# --------------------------------------------------------------------------
# Registration / routing
# --------------------------------------------------------------------------

View file

@ -17,6 +17,7 @@ from typing import Final
import httpx
import pytest
from openai import AsyncOpenAI, OpenAI
import litellm
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler
@ -69,6 +70,74 @@ def test_speech_makes_no_outbound_request(no_request_allowed):
assert no_request_allowed == []
@pytest.mark.parametrize("clinepass_key", [None, "cp-test-key"])
@pytest.mark.parametrize("custom_llm_provider", [None, "clinepass"])
@pytest.mark.parametrize("explicit_key", [None, "cp-explicit-key"])
def test_moderation_rejects_clinepass_before_credential_fallback(
monkeypatch, no_request_allowed, clinepass_key, custom_llm_provider, explicit_key
):
if clinepass_key is None:
monkeypatch.delenv("CLINEPASS_API_KEY", raising=False)
monkeypatch.setattr(litellm, "api_key", SENTINEL_OPENAI_KEY)
model: Final = "deepseek-v4-flash" if custom_llm_provider else "clinepass/deepseek-v4-flash"
with pytest.raises(litellm.BadRequestError, match="ClinePass does not support moderation endpoints"):
litellm.moderation(model=model, input="hi", custom_llm_provider=custom_llm_provider, api_key=explicit_key)
assert no_request_allowed == []
@pytest.mark.parametrize("clinepass_key", [None, "cp-test-key"])
@pytest.mark.parametrize("custom_llm_provider", [None, "clinepass"])
@pytest.mark.parametrize("explicit_key", [None, "cp-explicit-key"])
@pytest.mark.asyncio
async def test_async_moderation_rejects_clinepass_before_credential_fallback(
monkeypatch, no_request_allowed, clinepass_key, custom_llm_provider, explicit_key
):
if clinepass_key is None:
monkeypatch.delenv("CLINEPASS_API_KEY", raising=False)
monkeypatch.setattr(litellm, "api_key", SENTINEL_OPENAI_KEY)
model: Final = "deepseek-v4-flash" if custom_llm_provider else "clinepass/deepseek-v4-flash"
with pytest.raises(litellm.BadRequestError, match="ClinePass does not support moderation endpoints"):
await litellm.amoderation(
model=model, input="hi", custom_llm_provider=custom_llm_provider, api_key=explicit_key
)
assert no_request_allowed == []
@pytest.mark.parametrize("async_mode", [False, True])
@pytest.mark.asyncio
async def test_openai_moderation_remains_supported(async_mode):
sent = []
def respond(request):
sent.append((str(request.url), request.headers["authorization"]))
return httpx.Response(
200,
json={
"id": "modr-test",
"model": "moderation-test",
"results": [{"flagged": False, "categories": {"violence": False}, "category_scores": {"violence": 0}}],
},
)
transport: Final = httpx.MockTransport(respond)
if async_mode:
async with AsyncOpenAI(
api_key="sk-moderation-test", http_client=httpx.AsyncClient(transport=transport)
) as client:
response = await litellm.amoderation(model="openai/moderation-test", input="hi", client=client)
else:
with OpenAI(api_key="sk-moderation-test", http_client=httpx.Client(transport=transport)) as client:
response = litellm.moderation(model="moderation-test", input="hi", client=client)
assert sent == [("https://api.openai.com/v1/moderations", "Bearer sk-moderation-test")]
assert response.id == "modr-test"
assert response.results[0].flagged is False
def test_transcription_makes_no_outbound_request(no_request_allowed, tmp_path):
audio = tmp_path / "a.mp3"
audio.write_bytes(b"\x00\x00")