diff --git a/litellm/__init__.py b/litellm/__init__.py index e1da202b9ee..bab63f33366 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -1968,6 +1968,9 @@ if TYPE_CHECKING: from .llms.llamafile.chat.transformation import ( LlamafileChatConfig as _LlamafileChatConfig, ) + from .llms.llmman.chat.transformation import ( + LlmmanChatConfig as _LlmmanChatConfig, + ) from .llms.lm_studio.chat.transformation import ( LMStudioChatConfig as _LMStudioChatConfig, ) diff --git a/litellm/_lazy_imports_registry.py b/litellm/_lazy_imports_registry.py index aef3cbd9414..11219fa91c8 100644 --- a/litellm/_lazy_imports_registry.py +++ b/litellm/_lazy_imports_registry.py @@ -291,6 +291,7 @@ LLM_CONFIG_NAMES: Final = ( # Alias for backwards compatibility "VolcEngineConfig", # Alias for VolcEngineChatConfig "LlamafileChatConfig", + "LlmmanChatConfig", "LiteLLMProxyChatConfig", "VLLMConfig", "DeepSeekChatConfig", @@ -1140,6 +1141,10 @@ _LLM_CONFIGS_IMPORT_MAP: Final = { ".llms.llamafile.chat.transformation", "LlamafileChatConfig", ), + "LlmmanChatConfig": ( + ".llms.llmman.chat.transformation", + "LlmmanChatConfig", + ), "LiteLLMProxyChatConfig": ( ".llms.litellm_proxy.chat.transformation", "LiteLLMProxyChatConfig", diff --git a/litellm/constants.py b/litellm/constants.py index 39c10d71709..b470babeed2 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -739,6 +739,7 @@ LITELLM_CHAT_PROVIDERS: Final = [ "litellm_proxy", "hosted_vllm", "llamafile", + "llmman", "lm_studio", "galadriel", "gradient_ai", @@ -981,6 +982,7 @@ openai_compatible_providers: Final[list] = [ "litellm_proxy", "hosted_vllm", "llamafile", + "llmman", "lm_studio", "galadriel", "github_copilot", # GitHub Copilot Chat API @@ -1035,6 +1037,7 @@ openai_text_completion_compatible_providers: Final[list] = [ # providers that s "hosted_vllm", "meta_llama", "llamafile", + "llmman", "featherless_ai", "nebius", "dashscope", diff --git a/litellm/litellm_core_utils/get_llm_provider_logic.py b/litellm/litellm_core_utils/get_llm_provider_logic.py index d4642ae2aad..5706d9ac9a9 100644 --- a/litellm/litellm_core_utils/get_llm_provider_logic.py +++ b/litellm/litellm_core_utils/get_llm_provider_logic.py @@ -708,6 +708,11 @@ def _get_openai_compatible_provider_info( api_base, dynamic_api_key, ) = litellm.LlamafileChatConfig()._get_openai_compatible_provider_info(api_base, api_key) + elif custom_llm_provider == "llmman": + ( + api_base, + dynamic_api_key, + ) = litellm.LlmmanChatConfig()._get_openai_compatible_provider_info(api_base, api_key) elif custom_llm_provider == "datarobot": # DataRobot is OpenAI compatible. ( diff --git a/litellm/llms/llmman/chat/transformation.py b/litellm/llms/llmman/chat/transformation.py new file mode 100644 index 00000000000..31ce5cc7a2a --- /dev/null +++ b/litellm/llms/llmman/chat/transformation.py @@ -0,0 +1,18 @@ +from typing import Final + +from litellm.secret_managers.main import get_secret_str + +from ...openai.chat.gpt_transformation import OpenAIGPTConfig + +DEFAULT_API_BASE: Final = "http://127.0.0.1:17434/v1" +PLACEHOLDER_API_KEY: Final = "fake-api-key" + + +class LlmmanChatConfig(OpenAIGPTConfig): + def _get_openai_compatible_provider_info( + self, api_base: str | None, api_key: str | None + ) -> tuple[str | None, str | None]: + return ( + api_base or get_secret_str("LLMMAN_API_BASE") or DEFAULT_API_BASE, + api_key or get_secret_str("LLMMAN_API_KEY") or PLACEHOLDER_API_KEY, + ) diff --git a/litellm/main.py b/litellm/main.py index 6c85adf3ae8..8808f05c431 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -6585,6 +6585,7 @@ def embedding( elif ( custom_llm_provider == "openai_like" or custom_llm_provider == "llamafile" + or custom_llm_provider == "llmman" or custom_llm_provider == "lm_studio" ): api_base = api_base or litellm.api_base or get_secret_str("OPENAI_LIKE_API_BASE") diff --git a/litellm/proxy/public_endpoints/provider_create_fields.json b/litellm/proxy/public_endpoints/provider_create_fields.json index 67a8c356a4a..7b754bd5741 100644 --- a/litellm/proxy/public_endpoints/provider_create_fields.json +++ b/litellm/proxy/public_endpoints/provider_create_fields.json @@ -1995,6 +1995,34 @@ ], "default_model_placeholder": "gpt-3.5-turbo" }, + { + "provider": "LLMMAN", + "provider_display_name": "Llmman", + "litellm_provider": "llmman", + "credential_fields": [ + { + "key": "api_base", + "label": "API Base", + "placeholder": "http://127.0.0.1:17434/v1", + "tooltip": null, + "required": false, + "field_type": "text", + "options": null, + "default_value": null + }, + { + "key": "api_key", + "label": "API Key", + "placeholder": null, + "tooltip": null, + "required": false, + "field_type": "password", + "options": null, + "default_value": null + } + ], + "default_model_placeholder": "qwen3.8" + }, { "provider": "LM_STUDIO", "provider_display_name": "Lm Studio", diff --git a/litellm/types/utils.py b/litellm/types/utils.py index dba15bc99a5..ea753f6200c 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -4083,6 +4083,7 @@ class LlmProviders(str, Enum): HOSTED_VLLM = "hosted_vllm" TENCENT = "tencent" LLAMAFILE = "llamafile" + LLMMAN = "llmman" LM_STUDIO = "lm_studio" GALADRIEL = "galadriel" NEBIUS = "nebius" diff --git a/provider_endpoints_support.json b/provider_endpoints_support.json index 44ef9363b64..10b81064cb5 100644 --- a/provider_endpoints_support.json +++ b/provider_endpoints_support.json @@ -1547,6 +1547,24 @@ "interactions": true } }, + "llmman": { + "display_name": "Llmman (`llmman`)", + "url": "https://docs.litellm.ai/docs/providers/llmman", + "endpoints": { + "chat_completions": true, + "messages": true, + "responses": true, + "embeddings": true, + "image_generations": false, + "audio_transcriptions": false, + "audio_speech": false, + "moderations": false, + "batches": false, + "rerank": false, + "a2a": true, + "interactions": true + } + }, "lm_studio": { "display_name": "LM Studio (`lm_studio`)", "url": "https://docs.litellm.ai/docs/providers/lm_studio", diff --git a/tests/unit/llms/llmman/__init__.py b/tests/unit/llms/llmman/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/tests/unit/llms/llmman/chat/__init__.py b/tests/unit/llms/llmman/chat/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/tests/unit/llms/llmman/chat/test_llmman_chat_transformation.py b/tests/unit/llms/llmman/chat/test_llmman_chat_transformation.py new file mode 100644 index 00000000000..bb707ade4e9 --- /dev/null +++ b/tests/unit/llms/llmman/chat/test_llmman_chat_transformation.py @@ -0,0 +1,97 @@ +import json +from typing import Final +from unittest.mock import patch + +import httpx +import pytest +import respx + +import litellm +from litellm.llms.llmman.chat.transformation import LlmmanChatConfig + +DEFAULT_API_BASE: Final = "http://127.0.0.1:17434/v1" +ENV: Final = { + "LLMMAN_API_BASE": "https://env-api.example.com", + "LLMMAN_API_KEY": "env-key", +} + + +@pytest.mark.parametrize( + "api_base, api_key, env, expected_base, expected_key", + [ + ("https://user-api.example.com", "user-key", ENV, "https://user-api.example.com", "user-key"), + (None, None, ENV, "https://env-api.example.com", "env-key"), + (None, None, {}, DEFAULT_API_BASE, "fake-api-key"), + ("", "", ENV, "https://env-api.example.com", "env-key"), + ("", "", {}, DEFAULT_API_BASE, "fake-api-key"), + ("https://user-api.example.com", None, ENV, "https://user-api.example.com", "env-key"), + (None, "user-key", ENV, "https://env-api.example.com", "user-key"), + ], +) +def test_get_openai_compatible_provider_info(api_base, api_key, env, expected_base, expected_key): + with patch.dict("os.environ", env, clear=True): + assert LlmmanChatConfig()._get_openai_compatible_provider_info(api_base, api_key) == ( + expected_base, + expected_key, + ) + + +def test_get_llm_provider_routes_llmman_prefix(): + with patch.dict("os.environ", {}, clear=True): + result = litellm.get_llm_provider("llmman/my-custom-test-model") + + assert result == ("my-custom-test-model", "llmman", "fake-api-key", DEFAULT_API_BASE) + + +@respx.mock +def test_completion_hits_default_llmman_endpoint(): + route = respx.post(f"{DEFAULT_API_BASE}/chat/completions").mock( + return_value=httpx.Response( + 200, + json={ + "id": "chatcmpl-1", + "object": "chat.completion", + "created": 1, + "model": "my-custom-test-model", + "choices": [{"index": 0, "finish_reason": "stop", "message": {"role": "assistant", "content": "hi"}}], + "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}, + }, + ) + ) + + with patch.dict("os.environ", {}, clear=True): + response = litellm.completion( + model="llmman/my-custom-test-model", + messages=[{"role": "user", "content": "Hey, how's it going?"}], + max_tokens=100, + ) + + assert response.choices[0].message.content == "hi" + request = route.calls.last.request + assert request.headers["authorization"] == "Bearer fake-api-key" + body = json.loads(request.content) + assert body["model"] == "my-custom-test-model" + assert body["max_tokens"] == 100 + + +@respx.mock +def test_embedding_hits_default_llmman_endpoint(): + route = respx.post(f"{DEFAULT_API_BASE}/embeddings").mock( + return_value=httpx.Response( + 200, + json={ + "object": "list", + "data": [{"object": "embedding", "index": 0, "embedding": [0.1, 0.2]}], + "model": "my-embedding-model", + "usage": {"prompt_tokens": 1, "total_tokens": 1}, + }, + ) + ) + + with patch.dict("os.environ", {}, clear=True): + response = litellm.embedding(model="llmman/my-embedding-model", input=["hello"]) + + assert response.data[0]["embedding"] == [0.1, 0.2] + request = route.calls.last.request + assert request.headers["authorization"] == "Bearer fake-api-key" + assert json.loads(request.content)["model"] == "my-embedding-model"