diff --git a/litellm/llms/clinepass/chat/transformation.py b/litellm/llms/clinepass/chat/transformation.py index 6db0819a51a..c023335d876 100644 --- a/litellm/llms/clinepass/chat/transformation.py +++ b/litellm/llms/clinepass/chat/transformation.py @@ -14,7 +14,7 @@ ClinePass is OpenAI-compatible apart from two quirks, both handled here: Documentation: https://docs.cline.bot/ Credentials come only from the request's api_key or CLINEPASS_API_KEY. -Realtime endpoints are unsupported and rejected before HTTP dispatch. +Moderation and realtime endpoints are unsupported and rejected before dispatch. """ import json @@ -134,6 +134,18 @@ class ClinePassConfig(OpenAIGPTConfig): llm_provider="clinepass", ) + @staticmethod + def validate_moderation(model: str | None, custom_llm_provider: str | None = None) -> None: + if custom_llm_provider != "clinepass" and not (model or "").startswith("clinepass/"): + return + from litellm.exceptions import BadRequestError + + raise BadRequestError( + message="ClinePass does not support moderation endpoints", + model=model or "", + llm_provider="clinepass", + ) + def _get_openai_compatible_provider_info( self, api_base: str | None, api_key: str | None ) -> tuple[str | None, str | None]: diff --git a/litellm/main.py b/litellm/main.py index 83cf6b33d6b..aa1a632b0e7 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -7808,6 +7808,7 @@ def adapter_completion(*, adapter_id: str, **kwargs) -> BaseModel | AdapterCompl def moderation(input: str, model: str | None = None, api_key: str | None = None, **kwargs) -> OpenAIModerationResponse: + litellm.ClinePassConfig.validate_moderation(model=model, custom_llm_provider=kwargs.get("custom_llm_provider")) # only supports open ai for now api_key = api_key or litellm.api_key or litellm.openai_key or get_secret_str("OPENAI_API_KEY") @@ -7842,6 +7843,7 @@ async def amoderation( ) -> OpenAIModerationResponse: from openai import AsyncOpenAI + litellm.ClinePassConfig.validate_moderation(model=model, custom_llm_provider=custom_llm_provider) # only supports open ai for now api_key = api_key or litellm.api_key or litellm.openai_key or get_secret_str("OPENAI_API_KEY") optional_params: Final = GenericLiteLLMParams(**kwargs) diff --git a/litellm/responses/main.py b/litellm/responses/main.py index 14c12fc571d..b595ee731d7 100644 --- a/litellm/responses/main.py +++ b/litellm/responses/main.py @@ -2405,10 +2405,13 @@ async def _aresponses_websocket( resolved_api_key: Final = ( dynamic_api_key + or api_key or litellm_params.api_key - or litellm.api_key - or litellm.openai_key - or get_secret_str("OPENAI_API_KEY") + or ( + (litellm.api_key or litellm.openai_key or get_secret_str("OPENAI_API_KEY")) + if responses_api_provider_config is not None + else None + ) ) # Extract params that we're passing explicitly to avoid duplicates in **kwargs diff --git a/litellm/responses/streaming_iterator.py b/litellm/responses/streaming_iterator.py index 5e045c3e84f..475acd5b967 100644 --- a/litellm/responses/streaming_iterator.py +++ b/litellm/responses/streaming_iterator.py @@ -2824,9 +2824,10 @@ class ManagedResponsesWebSocketHandler: def _inject_credentials(self, call_kwargs: dict[str, object], model: str | None = None) -> None: """Inject connection-level credentials and metadata into call_kwargs.""" - if self.api_key is not None: + same_provider: Final = self._same_provider(model) + if self.api_key is not None and same_provider: call_kwargs["api_key"] = self.api_key - if self.api_base is not None: + if self.api_base is not None and same_provider: call_kwargs["api_base"] = self.api_base if self.timeout is not None: call_kwargs["timeout"] = self.timeout @@ -2835,7 +2836,7 @@ class ManagedResponsesWebSocketHandler: # (e.g., connection is vertex_ai but event says openai/gpt-4), let litellm # re-resolve from the model string. Same-provider model variants (e.g., # vertex_ai/gemini-2.0 -> vertex_ai/gemini-1.5) still inherit the provider. - if self.custom_llm_provider is not None and self._same_provider(model): + if self.custom_llm_provider is not None and same_provider: call_kwargs["custom_llm_provider"] = self.custom_llm_provider if self.litellm_metadata: call_kwargs["litellm_metadata"] = dict(self.litellm_metadata) diff --git a/tests/unit/llms/clinepass/chat/test_clinepass_chat_transformation.py b/tests/unit/llms/clinepass/chat/test_clinepass_chat_transformation.py index e1fc3a2bf3f..85115d754a2 100644 --- a/tests/unit/llms/clinepass/chat/test_clinepass_chat_transformation.py +++ b/tests/unit/llms/clinepass/chat/test_clinepass_chat_transformation.py @@ -13,14 +13,17 @@ from unittest.mock import patch import httpx import pytest +from starlette.websockets import WebSocket import litellm +from litellm.litellm_core_utils.litellm_logging import Logging from litellm.llms.clinepass.chat.transformation import ( ClinePassConfig, _apply_model_prefix, _unwrap_response_envelope, ) from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler +from litellm.responses.main import _aresponses_websocket from litellm.types.utils import LlmProviders from litellm.utils import ProviderConfigManager @@ -122,6 +125,89 @@ async def test_async_chat_sends_only_clinepass_credentials(monkeypatch, unrelate assert response.choices[0].message.content == "pong" +@pytest.mark.parametrize("credential_source", ["missing", "environment", "explicit"]) +@pytest.mark.parametrize("connection_provider", ["clinepass", "mistral"]) +@pytest.mark.asyncio +async def test_managed_responses_websocket_sends_only_clinepass_credentials( + monkeypatch, unrelated_credentials, credential_source, connection_provider +): + if credential_source == "missing": + monkeypatch.delenv("CLINEPASS_API_KEY", raising=False) + connection_key: Final = ( + "sk-unrelated-mistral" + if connection_provider == "mistral" + else "cp-request-key" + if credential_source == "explicit" + else None + ) + sent = [] + received = [] + lifecycle = iter(({"type": "websocket.connect"}, {"type": "websocket.disconnect", "code": 1000})) + + async def receive(): + return next(lifecycle) + + async def send(message): + if message["type"] == "websocket.send": + received.append(json.loads(message["text"])) + + websocket: Final = WebSocket( + scope={"type": "websocket", "path": "/v1/responses", "headers": [], "query_string": b""}, + receive=receive, + send=send, + ) + await websocket.accept() + chunk: Final = { + "id": "chatcmpl-test", + "object": "chat.completion.chunk", + "created": 1, + "model": "cline-pass/deepseek-v4-flash", + "choices": [{"index": 0, "delta": {"role": "assistant", "content": "pong"}, "finish_reason": "stop"}], + } + + async def fake_post(self, url, *args, **kwargs): + sent.append((str(url), httpx.Headers(kwargs["headers"]).get("authorization"))) + return httpx.Response( + 200, + content=f"data: {json.dumps(chunk)}\n\ndata: [DONE]\n\n".encode(), + headers={"content-type": "text/event-stream"}, + request=httpx.Request("POST", str(url)), + ) + + with patch.object(AsyncHTTPHandler, "post", fake_post): + result = await _aresponses_websocket.__wrapped__( + model=f"{connection_provider}/deepseek-v4-flash", + websocket=websocket, + api_key=connection_key, + first_message=json.dumps( + {"type": "response.create", "model": "clinepass/deepseek-v4-flash", "input": "ping"} + ), + litellm_logging_obj=Logging( + model=f"{connection_provider}/deepseek-v4-flash", + messages=[], + stream=True, + call_type="aresponses", + start_time=0, + litellm_call_id="cp-ws-test", + function_id="cp-ws-test", + ), + ) + + expected_key: Final = ( + connection_key + if connection_provider == "clinepass" and credential_source == "explicit" + else API_KEY + if credential_source != "missing" + else None + ) + assert sent == [ + ("https://api.cline.bot/api/v1/chat/completions", f"Bearer {expected_key}" if expected_key else None) + ] + assert result is None + assert "response.completed" in [event["type"] for event in received] + assert "error" not in [event["type"] for event in received] + + # -------------------------------------------------------------------------- # Registration / routing # -------------------------------------------------------------------------- diff --git a/tests/unit/llms/clinepass/test_clinepass_endpoint_guard.py b/tests/unit/llms/clinepass/test_clinepass_endpoint_guard.py index 35c490db314..0a06a15cc1a 100644 --- a/tests/unit/llms/clinepass/test_clinepass_endpoint_guard.py +++ b/tests/unit/llms/clinepass/test_clinepass_endpoint_guard.py @@ -17,6 +17,7 @@ from typing import Final import httpx import pytest +from openai import AsyncOpenAI, OpenAI import litellm from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler @@ -69,6 +70,74 @@ def test_speech_makes_no_outbound_request(no_request_allowed): assert no_request_allowed == [] +@pytest.mark.parametrize("clinepass_key", [None, "cp-test-key"]) +@pytest.mark.parametrize("custom_llm_provider", [None, "clinepass"]) +@pytest.mark.parametrize("explicit_key", [None, "cp-explicit-key"]) +def test_moderation_rejects_clinepass_before_credential_fallback( + monkeypatch, no_request_allowed, clinepass_key, custom_llm_provider, explicit_key +): + if clinepass_key is None: + monkeypatch.delenv("CLINEPASS_API_KEY", raising=False) + monkeypatch.setattr(litellm, "api_key", SENTINEL_OPENAI_KEY) + model: Final = "deepseek-v4-flash" if custom_llm_provider else "clinepass/deepseek-v4-flash" + + with pytest.raises(litellm.BadRequestError, match="ClinePass does not support moderation endpoints"): + litellm.moderation(model=model, input="hi", custom_llm_provider=custom_llm_provider, api_key=explicit_key) + + assert no_request_allowed == [] + + +@pytest.mark.parametrize("clinepass_key", [None, "cp-test-key"]) +@pytest.mark.parametrize("custom_llm_provider", [None, "clinepass"]) +@pytest.mark.parametrize("explicit_key", [None, "cp-explicit-key"]) +@pytest.mark.asyncio +async def test_async_moderation_rejects_clinepass_before_credential_fallback( + monkeypatch, no_request_allowed, clinepass_key, custom_llm_provider, explicit_key +): + if clinepass_key is None: + monkeypatch.delenv("CLINEPASS_API_KEY", raising=False) + monkeypatch.setattr(litellm, "api_key", SENTINEL_OPENAI_KEY) + model: Final = "deepseek-v4-flash" if custom_llm_provider else "clinepass/deepseek-v4-flash" + + with pytest.raises(litellm.BadRequestError, match="ClinePass does not support moderation endpoints"): + await litellm.amoderation( + model=model, input="hi", custom_llm_provider=custom_llm_provider, api_key=explicit_key + ) + + assert no_request_allowed == [] + + +@pytest.mark.parametrize("async_mode", [False, True]) +@pytest.mark.asyncio +async def test_openai_moderation_remains_supported(async_mode): + sent = [] + + def respond(request): + sent.append((str(request.url), request.headers["authorization"])) + return httpx.Response( + 200, + json={ + "id": "modr-test", + "model": "moderation-test", + "results": [{"flagged": False, "categories": {"violence": False}, "category_scores": {"violence": 0}}], + }, + ) + + transport: Final = httpx.MockTransport(respond) + if async_mode: + async with AsyncOpenAI( + api_key="sk-moderation-test", http_client=httpx.AsyncClient(transport=transport) + ) as client: + response = await litellm.amoderation(model="openai/moderation-test", input="hi", client=client) + else: + with OpenAI(api_key="sk-moderation-test", http_client=httpx.Client(transport=transport)) as client: + response = litellm.moderation(model="moderation-test", input="hi", client=client) + + assert sent == [("https://api.openai.com/v1/moderations", "Bearer sk-moderation-test")] + assert response.id == "modr-test" + assert response.results[0].flagged is False + + def test_transcription_makes_no_outbound_request(no_request_allowed, tmp_path): audio = tmp_path / "a.mp3" audio.write_bytes(b"\x00\x00")