diff --git a/litellm/__init__.py b/litellm/__init__.py index c8df4394a06..0ff58106942 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -1954,6 +1954,9 @@ if TYPE_CHECKING: from .llms.llamafile.chat.transformation import ( LlamafileChatConfig as _LlamafileChatConfig, ) + from .llms.llmman.chat.transformation import ( + LlmmanChatConfig as _LlmmanChatConfig, + ) from .llms.lm_studio.chat.transformation import ( LMStudioChatConfig as _LMStudioChatConfig, ) diff --git a/litellm/_lazy_imports_registry.py b/litellm/_lazy_imports_registry.py index 9a53273c9d5..7dadcacfb0a 100644 --- a/litellm/_lazy_imports_registry.py +++ b/litellm/_lazy_imports_registry.py @@ -289,6 +289,7 @@ LLM_CONFIG_NAMES: Final = ( # Alias for backwards compatibility "VolcEngineConfig", # Alias for VolcEngineChatConfig "LlamafileChatConfig", + "LlmmanChatConfig", "LiteLLMProxyChatConfig", "VLLMConfig", "DeepSeekChatConfig", @@ -1135,6 +1136,10 @@ _LLM_CONFIGS_IMPORT_MAP: Final = { ".llms.llamafile.chat.transformation", "LlamafileChatConfig", ), + "LlmmanChatConfig": ( + ".llms.llmman.chat.transformation", + "LlmmanChatConfig", + ), "LiteLLMProxyChatConfig": ( ".llms.litellm_proxy.chat.transformation", "LiteLLMProxyChatConfig", diff --git a/litellm/constants.py b/litellm/constants.py index e5b662bd515..258ad1babc7 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -729,6 +729,7 @@ LITELLM_CHAT_PROVIDERS: Final = [ "litellm_proxy", "hosted_vllm", "llamafile", + "llmman", "lm_studio", "galadriel", "gradient_ai", @@ -968,6 +969,7 @@ openai_compatible_providers: Final[list] = [ "litellm_proxy", "hosted_vllm", "llamafile", + "llmman", "lm_studio", "galadriel", "github_copilot", # GitHub Copilot Chat API @@ -1020,6 +1022,7 @@ openai_text_completion_compatible_providers: Final[list] = [ # providers that s "hosted_vllm", "meta_llama", "llamafile", + "llmman", "featherless_ai", "nebius", "dashscope", diff --git a/litellm/litellm_core_utils/get_llm_provider_logic.py b/litellm/litellm_core_utils/get_llm_provider_logic.py index f622098920c..172ae213e65 100644 --- a/litellm/litellm_core_utils/get_llm_provider_logic.py +++ b/litellm/litellm_core_utils/get_llm_provider_logic.py @@ -680,6 +680,12 @@ def _get_openai_compatible_provider_info( api_base, dynamic_api_key, ) = litellm.LlamafileChatConfig()._get_openai_compatible_provider_info(api_base, api_key) + elif custom_llm_provider == "llmman": + # llmman is OpenAI compatible. + ( + api_base, + dynamic_api_key, + ) = litellm.LlmmanChatConfig()._get_openai_compatible_provider_info(api_base, api_key) elif custom_llm_provider == "datarobot": # DataRobot is OpenAI compatible. ( diff --git a/litellm/llms/llmman/chat/transformation.py b/litellm/llms/llmman/chat/transformation.py new file mode 100644 index 00000000000..0351a44b828 --- /dev/null +++ b/litellm/llms/llmman/chat/transformation.py @@ -0,0 +1,43 @@ +from typing import Final + +from litellm.secret_managers.main import get_secret_str + +from ...openai.chat.gpt_transformation import OpenAIGPTConfig + + +class LlmmanChatConfig(OpenAIGPTConfig): + """Configuration for llmman's OpenAI-compatible chat API. + + llmman is a local model runner that serves OpenAI-, Ollama- and + Anthropic-compatible APIs. See https://github.com/llmmanorg/llmman + """ + + @staticmethod + def _resolve_api_key(api_key: str | None = None) -> str: + """Resolve the API key, preferring the user-provided value over + ``LLMMAN_API_KEY``. + + Returns a placeholder when neither is set: llmman does not require a + key, but the underlying OpenAI library expects a non-None value. + """ + return api_key or get_secret_str("LLMMAN_API_KEY") or "fake-api-key" + + @staticmethod + def _resolve_api_base(api_base: str | None = None) -> str | None: + """Resolve the API base, preferring the user-provided value over + ``LLMMAN_API_BASE``, then falling back to the default `llmman serve` + address. + + See: https://github.com/llmmanorg/llmman#serve + """ + return ( + api_base or get_secret_str("LLMMAN_API_BASE") or "http://127.0.0.1:17434/v1" + ) + + def _get_openai_compatible_provider_info( + self, api_base: str | None, api_key: str | None + ) -> tuple[str | None, str | None]: + api_base = LlmmanChatConfig._resolve_api_base(api_base) + dynamic_api_key: Final = LlmmanChatConfig._resolve_api_key(api_key) + + return api_base, dynamic_api_key diff --git a/litellm/main.py b/litellm/main.py index 7f4b34d28a0..1b0237cb307 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -6544,6 +6544,7 @@ def embedding( elif ( custom_llm_provider == "openai_like" or custom_llm_provider == "llamafile" + or custom_llm_provider == "llmman" or custom_llm_provider == "lm_studio" ): api_base = api_base or litellm.api_base or get_secret_str("OPENAI_LIKE_API_BASE") diff --git a/litellm/types/utils.py b/litellm/types/utils.py index dfc98a9d89d..cf5608b5261 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -4236,6 +4236,7 @@ class LlmProviders(str, Enum): HOSTED_VLLM = "hosted_vllm" TENCENT = "tencent" LLAMAFILE = "llamafile" + LLMMAN = "llmman" LM_STUDIO = "lm_studio" GALADRIEL = "galadriel" NEBIUS = "nebius" diff --git a/tests/test_litellm/llms/llmman/chat/test_llmman_chat_transformation.py b/tests/test_litellm/llms/llmman/chat/test_llmman_chat_transformation.py new file mode 100644 index 00000000000..c3cc4d6e865 --- /dev/null +++ b/tests/test_litellm/llms/llmman/chat/test_llmman_chat_transformation.py @@ -0,0 +1,179 @@ +from typing import Optional +from unittest.mock import patch + +import pytest + +import litellm +from litellm.llms.llmman.chat.transformation import LlmmanChatConfig + + +@pytest.mark.parametrize( + "input_api_key, env_api_key, expected_api_key", + [ + ("user-provided-key", "secret-key", "user-provided-key"), + (None, "secret-key", "secret-key"), + (None, None, "fake-api-key"), + ("", "secret-key", "secret-key"), # Empty string should fall back to secret + ( + "", + None, + "fake-api-key", + ), # Empty string with no secret should use the fake key + ], +) +def test_resolve_api_key(input_api_key, env_api_key, expected_api_key): + env = {} + if env_api_key is not None: + env["LLMMAN_API_KEY"] = env_api_key + + with patch.dict("os.environ", env, clear=True): + result = LlmmanChatConfig._resolve_api_key(input_api_key) + assert result == expected_api_key + + +@pytest.mark.parametrize( + "input_api_base, env_api_base, expected_api_base", + [ + ( + "https://user-api.example.com", + "https://secret-api.example.com", + "https://user-api.example.com", + ), + ( + None, + "https://secret-api.example.com", + "https://secret-api.example.com", + ), + (None, None, "http://127.0.0.1:17434/v1"), + ( + "", + "https://secret-api.example.com", + "https://secret-api.example.com", + ), # Empty string should fall back + ], +) +def test_resolve_api_base( + input_api_base, + env_api_base, + expected_api_base, +): + env = {} + if env_api_base is not None: + env["LLMMAN_API_BASE"] = env_api_base + + with patch.dict("os.environ", env, clear=True): + result = LlmmanChatConfig._resolve_api_base(input_api_base) + assert result == expected_api_base + + +@pytest.mark.parametrize( + "api_base, api_key, env_base, env_key, expected_base, expected_key", + [ + # User-provided values + ( + "https://user-api.example.com", + "user-key", + "https://secret-api.example.com", + "secret-key", + "https://user-api.example.com", + "user-key", + ), + # Fallback to env vars + ( + None, + None, + "https://secret-api.example.com", + "secret-key", + "https://secret-api.example.com", + "secret-key", + ), + # Nothing provided, use defaults + (None, None, None, None, "http://127.0.0.1:17434/v1", "fake-api-key"), + # Mixed scenarios + ( + "https://user-api.example.com", + None, + None, + "secret-key", + "https://user-api.example.com", + "secret-key", + ), + ( + None, + "user-key", + "https://secret-api.example.com", + None, + "https://secret-api.example.com", + "user-key", + ), + ], +) +def test_get_openai_compatible_provider_info( + api_base, api_key, env_base, env_key, expected_base, expected_key +): + config = LlmmanChatConfig() + + env = {} + if env_base is not None: + env["LLMMAN_API_BASE"] = env_base + if env_key is not None: + env["LLMMAN_API_KEY"] = env_key + + patch_base = patch.object( + LlmmanChatConfig, + "_resolve_api_base", + wraps=LlmmanChatConfig._resolve_api_base, + ) + patch_key = patch.object( + LlmmanChatConfig, + "_resolve_api_key", + wraps=LlmmanChatConfig._resolve_api_key, + ) + + with ( + patch.dict("os.environ", env, clear=True), + patch_base as mock_base, + patch_key as mock_key, + ): + result_base, result_key = config._get_openai_compatible_provider_info( + api_base, api_key + ) + + assert result_base == expected_base + assert result_key == expected_key + + mock_base.assert_called_once_with(api_base) + mock_key.assert_called_once_with(api_key) + + +def test_completion_with_custom_llmman_model(): + with patch( + "litellm.main.openai_chat_completions.completion" + ) as mock_llmman_completion_func: + mock_llmman_completion_func.return_value = ( + {} + ) # Return an empty dictionary for the mocked response + + provider = "llmman" + model_name = "my-custom-test-model" + model = f"{provider}/{model_name}" + messages = [{"role": "user", "content": "Hey, how's it going?"}] + + _ = litellm.completion( + model=model, + messages=messages, + max_retries=2, + max_tokens=100, + ) + + mock_llmman_completion_func.assert_called_once() + _, call_kwargs = mock_llmman_completion_func.call_args + assert call_kwargs.get("custom_llm_provider") == provider + assert call_kwargs.get("model") == model_name + assert call_kwargs.get("messages") == messages + assert call_kwargs.get("api_base") == "http://127.0.0.1:17434/v1" + assert call_kwargs.get("api_key") == "fake-api-key" + optional_params = call_kwargs.get("optional_params") + assert optional_params + assert optional_params.get("max_retries") == 2 + assert optional_params.get("max_tokens") == 100