From 9ceb46f937579a95cd10dcdc6aaac551602a3649 Mon Sep 17 00:00:00 2001 From: Bhagirath Mehta Date: Mon, 1 Jun 2026 13:45:49 -0500 Subject: [PATCH] Add Foundry Local provider and docs Add Foundry Local as a first-class OpenAI-compatible provider, wire it into the dashboard and proxy metadata, include the official upstream logo, and document the current SDK-managed local endpoint flow for Foundry Local. Files changed: litellm/llms/foundry_local/, litellm/utils.py, litellm/litellm_core_utils/get_llm_provider_logic.py, tests/test_litellm/llms/foundry_local/test_foundry_local_chat_transformation.py, docs/my-website/docs/providers/foundry_local.md, provider_endpoints_support.json, ui/litellm-dashboard/src/components/provider_info_helpers.tsx, litellm/proxy/public_endpoints/provider_create_fields.json Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- .../docs/providers/foundry_local.md | 222 ++++++++++++++++++ litellm/__init__.py | 4 + litellm/_lazy_imports_registry.py | 5 + litellm/constants.py | 2 + .../get_llm_provider_logic.py | 8 + litellm/llms/foundry_local/__init__.py | 0 litellm/llms/foundry_local/chat/__init__.py | 0 .../llms/foundry_local/chat/transformation.py | 23 ++ .../out/assets/logos/foundry_local.svg | 40 ++++ .../provider_create_fields.json | 28 +++ litellm/types/utils.py | 1 + litellm/utils.py | 4 + provider_endpoints_support.json | 18 ++ .../test_foundry_local_chat_transformation.py | 88 +++++++ .../public/assets/logos/foundry_local.svg | 40 ++++ .../src/components/provider_info_helpers.tsx | 3 + 16 files changed, 486 insertions(+) create mode 100644 docs/my-website/docs/providers/foundry_local.md create mode 100644 litellm/llms/foundry_local/__init__.py create mode 100644 litellm/llms/foundry_local/chat/__init__.py create mode 100644 litellm/llms/foundry_local/chat/transformation.py create mode 100644 litellm/proxy/_experimental/out/assets/logos/foundry_local.svg create mode 100644 tests/test_litellm/llms/foundry_local/test_foundry_local_chat_transformation.py create mode 100644 ui/litellm-dashboard/public/assets/logos/foundry_local.svg diff --git a/docs/my-website/docs/providers/foundry_local.md b/docs/my-website/docs/providers/foundry_local.md new file mode 100644 index 00000000000..be71b3ae406 --- /dev/null +++ b/docs/my-website/docs/providers/foundry_local.md @@ -0,0 +1,222 @@ +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; + +# Foundry Local + +https://github.com/microsoft/Foundry-Local + +:::tip + +**We support ALL Foundry Local models, just set `model=foundry_local/` as a prefix when sending litellm requests** + +::: + + +| Property | Details | +|-------|-------| +| Description | Run AI models on-device with Microsoft Foundry Local. | +| Provider Route on LiteLLM | `foundry_local/` | +| Provider Doc | [Foundry Local ↗](https://learn.microsoft.com/en-us/azure/foundry-local/) | +| Supported OpenAI Endpoints | `/chat/completions` | + +## Quick Start + +Foundry Local itself is SDK-first and does **not** need to run as a web server for in-process apps. LiteLLM uses the optional OpenAI-compatible web service, so point `FOUNDRY_LOCAL_API_BASE` at `manager.urls[0]/v1`. + + + + +```bash +# Windows (recommended for hardware acceleration) +pip install foundry-local-sdk-winml + +# macOS/Linux +pip install foundry-local-sdk +``` + +```python +import os +from foundry_local_sdk import Configuration, FoundryLocalManager + +FoundryLocalManager.initialize(Configuration(app_name="litellm-foundry-local")) +manager = FoundryLocalManager.instance + +model = manager.catalog.get_model("qwen2.5-0.5b") +model.download() +model.load() +manager.start_web_service() + +os.environ["FOUNDRY_LOCAL_API_BASE"] = f"{manager.urls[0].rstrip('/')}/v1" +print(os.environ["FOUNDRY_LOCAL_API_BASE"]) +``` + + + + +```bash +# Windows (recommended for hardware acceleration) +npm install foundry-local-sdk-winml + +# macOS/Linux +npm install foundry-local-sdk +``` + +```javascript +import { FoundryLocalManager } from "foundry-local-sdk"; + +const manager = FoundryLocalManager.create({ + appName: "litellm-foundry-local", +}); + +const model = await manager.catalog.getModel("qwen2.5-0.5b"); +await model.download(); +await model.load(); +manager.startWebService(); + +const apiBase = `${manager.urls[0].replace(/\/$/, "")}/v1`; +console.log(apiBase); +``` + + + + +:::tip + +Foundry Local 1.2 also adds cancellable model and EP downloads (`threading.Event` in Python, `AbortController` in JavaScript) and serves the OpenAI Responses API from the same `/v1` endpoint. LiteLLM currently uses `/chat/completions`, so the local server is optional in general but required for this HTTP integration path. + +::: + +## API Key +```python +# env variable +os.environ['FOUNDRY_LOCAL_API_BASE'] # e.g. http://127.0.0.1:5272/v1 +os.environ['FOUNDRY_LOCAL_API_KEY'] # optional, not required for local use +``` + +## Sample Usage +```python +from litellm import completion +import os + +os.environ['FOUNDRY_LOCAL_API_BASE'] = "http://127.0.0.1:5272/v1" + +response = completion( + model="foundry_local/qwen2.5-0.5b", + messages=[ + { + "role": "user", + "content": "What's the weather like in Boston today in Fahrenheit?", + } + ] +) +print(response) +``` + +## Sample Usage - Streaming +```python +from litellm import completion +import os + +os.environ['FOUNDRY_LOCAL_API_BASE'] = "http://127.0.0.1:5272/v1" + +response = completion( + model="foundry_local/qwen2.5-0.5b", + messages=[ + { + "role": "user", + "content": "What's the weather like in Boston today in Fahrenheit?", + } + ], + stream=True, +) + +for chunk in response: + print(chunk) +``` + +## Usage with DSPy + +Foundry Local works seamlessly with [DSPy](https://dspy.ai/) via LiteLLM: + +```python +import dspy +import os + +os.environ['FOUNDRY_LOCAL_API_BASE'] = "http://127.0.0.1:5272/v1" + +lm = dspy.LM("foundry_local/qwen2.5-0.5b") +dspy.configure(lm=lm) +``` + +## Usage with LiteLLM Proxy Server + +Here's how to call a Foundry Local model with the LiteLLM Proxy Server + +1. Modify the config.yaml + + ```yaml + model_list: + - model_name: my-model + litellm_params: + model: foundry_local/qwen2.5-0.5b + api_base: http://127.0.0.1:5272/v1 + ``` + + +2. Start the proxy + + ```bash + $ litellm --config /path/to/config.yaml + ``` + +3. Send Request to LiteLLM Proxy Server + + + + + + ```python + import openai + client = openai.OpenAI( + api_key="sk-1234", # pass litellm proxy key, if you're using virtual keys + base_url="http://0.0.0.0:4000" # litellm-proxy-base url + ) + + response = client.chat.completions.create( + model="my-model", + messages = [ + { + "role": "user", + "content": "what llm are you" + } + ], + ) + + print(response) + ``` + + + + + ```shell + curl --location 'http://0.0.0.0:4000/chat/completions' \ + --header 'Authorization: Bearer sk-1234' \ + --header 'Content-Type: application/json' \ + --data '{ + "model": "my-model", + "messages": [ + { + "role": "user", + "content": "what llm are you" + } + ], + }' + ``` + + + + + +## Supported Parameters + +See [Supported Parameters](../completion/input.md#translated-openai-params) for supported parameters. diff --git a/litellm/__init__.py b/litellm/__init__.py index 56d516536e8..8b3a0dd7c7f 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -1803,6 +1803,9 @@ if TYPE_CHECKING: from .llms.lm_studio.chat.transformation import ( LMStudioChatConfig as _LMStudioChatConfig, ) + from .llms.foundry_local.chat.transformation import ( + FoundryLocalChatConfig as _FoundryLocalChatConfig, + ) from .llms.lm_studio.embed.transformation import ( LmStudioEmbeddingConfig as _LmStudioEmbeddingConfig, ) @@ -1827,6 +1830,7 @@ if TYPE_CHECKING: DeepInfraConfig: Type[_DeepInfraConfig] LlamafileChatConfig: Type[_LlamafileChatConfig] LMStudioChatConfig: Type[_LMStudioChatConfig] + FoundryLocalChatConfig: Type[_FoundryLocalChatConfig] LmStudioEmbeddingConfig: Type[_LmStudioEmbeddingConfig] IBMWatsonXEmbeddingConfig: Type[_IBMWatsonXEmbeddingConfig] VertexAIConfig: Type[_VertexGeminiConfig] # Alias for VertexGeminiConfig diff --git a/litellm/_lazy_imports_registry.py b/litellm/_lazy_imports_registry.py index 17eb6609292..b04244ba4f1 100644 --- a/litellm/_lazy_imports_registry.py +++ b/litellm/_lazy_imports_registry.py @@ -283,6 +283,7 @@ LLM_CONFIG_NAMES = ( "VLLMConfig", "DeepSeekChatConfig", "LMStudioChatConfig", + "FoundryLocalChatConfig", "LmStudioEmbeddingConfig", "NscaleConfig", "PerplexityChatConfig", @@ -1083,6 +1084,10 @@ _LLM_CONFIGS_IMPORT_MAP = { "VLLMConfig": (".llms.vllm.completion.transformation", "VLLMConfig"), "DeepSeekChatConfig": (".llms.deepseek.chat.transformation", "DeepSeekChatConfig"), "LMStudioChatConfig": (".llms.lm_studio.chat.transformation", "LMStudioChatConfig"), + "FoundryLocalChatConfig": ( + ".llms.foundry_local.chat.transformation", + "FoundryLocalChatConfig", + ), "LmStudioEmbeddingConfig": ( ".llms.lm_studio.embed.transformation", "LmStudioEmbeddingConfig", diff --git a/litellm/constants.py b/litellm/constants.py index ae98b37d6e6..d8f11eebb0f 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -603,6 +603,7 @@ LITELLM_CHAT_PROVIDERS = [ "hosted_vllm", "llamafile", "lm_studio", + "foundry_local", "galadriel", "gradient_ai", "github_copilot", # GitHub Copilot Chat API @@ -813,6 +814,7 @@ openai_compatible_providers: List = [ "hosted_vllm", "llamafile", "lm_studio", + "foundry_local", "galadriel", "github_copilot", # GitHub Copilot Chat API "chatgpt", # ChatGPT subscription API diff --git a/litellm/litellm_core_utils/get_llm_provider_logic.py b/litellm/litellm_core_utils/get_llm_provider_logic.py index ba6d438f16c..c73e30b9dc2 100644 --- a/litellm/litellm_core_utils/get_llm_provider_logic.py +++ b/litellm/litellm_core_utils/get_llm_provider_logic.py @@ -731,6 +731,14 @@ def _get_openai_compatible_provider_info( # noqa: PLR0915 ) = litellm.LMStudioChatConfig()._get_openai_compatible_provider_info( api_base, api_key ) + elif custom_llm_provider == "foundry_local": + # foundry_local is openai compatible, we just need to set this to custom_openai + ( + api_base, + dynamic_api_key, + ) = litellm.FoundryLocalChatConfig()._get_openai_compatible_provider_info( + api_base, api_key + ) elif custom_llm_provider == "deepseek": # deepseek is openai compatible, we just need to set this to custom_openai and have the api_base be https://api.deepseek.com/v1 api_base = ( diff --git a/litellm/llms/foundry_local/__init__.py b/litellm/llms/foundry_local/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/litellm/llms/foundry_local/chat/__init__.py b/litellm/llms/foundry_local/chat/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/litellm/llms/foundry_local/chat/transformation.py b/litellm/llms/foundry_local/chat/transformation.py new file mode 100644 index 00000000000..2e3220fab2f --- /dev/null +++ b/litellm/llms/foundry_local/chat/transformation.py @@ -0,0 +1,23 @@ +""" +Translate from OpenAI's `/v1/chat/completions` to Foundry Local's `/v1/chat/completions` + +Foundry Local (https://github.com/microsoft/Foundry-Local) runs LLMs on-device +and exposes an OpenAI-compatible REST API endpoint. +""" + +from typing import Optional, Tuple + +from litellm.secret_managers.main import get_secret_str + +from ...openai.chat.gpt_transformation import OpenAIGPTConfig + + +class FoundryLocalChatConfig(OpenAIGPTConfig): + def _get_openai_compatible_provider_info( + self, api_base: Optional[str], api_key: Optional[str] + ) -> Tuple[Optional[str], Optional[str]]: + api_base = api_base or get_secret_str("FOUNDRY_LOCAL_API_BASE") # type: ignore + dynamic_api_key = ( + api_key or get_secret_str("FOUNDRY_LOCAL_API_KEY") or "fake-api-key" + ) # Foundry Local does not require an api key + return api_base, dynamic_api_key diff --git a/litellm/proxy/_experimental/out/assets/logos/foundry_local.svg b/litellm/proxy/_experimental/out/assets/logos/foundry_local.svg new file mode 100644 index 00000000000..412a6fb70d6 --- /dev/null +++ b/litellm/proxy/_experimental/out/assets/logos/foundry_local.svg @@ -0,0 +1,40 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/litellm/proxy/public_endpoints/provider_create_fields.json b/litellm/proxy/public_endpoints/provider_create_fields.json index 163c9648de7..196f984259f 100644 --- a/litellm/proxy/public_endpoints/provider_create_fields.json +++ b/litellm/proxy/public_endpoints/provider_create_fields.json @@ -1618,6 +1618,34 @@ ], "default_model_placeholder": "gpt-3.5-turbo" }, + { + "provider": "FOUNDRY_LOCAL", + "provider_display_name": "Foundry Local", + "litellm_provider": "foundry_local", + "credential_fields": [ + { + "key": "api_base", + "label": "API Base", + "placeholder": "http://localhost:5272/v1", + "tooltip": "Foundry Local endpoint (from foundry-local-sdk or manually set)", + "required": false, + "field_type": "text", + "options": null, + "default_value": null + }, + { + "key": "api_key", + "label": "API Key", + "placeholder": null, + "tooltip": "Optional - Foundry Local does not require authentication", + "required": false, + "field_type": "password", + "options": null, + "default_value": null + } + ], + "default_model_placeholder": "phi-3.5-mini" + }, { "provider": "MARITALK", "provider_display_name": "Maritalk", diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 5574d616fac..93dcba8a0ba 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -3317,6 +3317,7 @@ class LlmProviders(str, Enum): HOSTED_VLLM = "hosted_vllm" LLAMAFILE = "llamafile" LM_STUDIO = "lm_studio" + FOUNDRY_LOCAL = "foundry_local" GALADRIEL = "galadriel" NEBIUS = "nebius" INFINITY = "infinity" diff --git a/litellm/utils.py b/litellm/utils.py index 5a9dccc089e..7a85d057e5e 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -8272,6 +8272,10 @@ class ProviderConfigManager: LlmProviders.HOSTED_VLLM: (lambda: litellm.HostedVLLMChatConfig(), False), LlmProviders.LLAMAFILE: (lambda: litellm.LlamafileChatConfig(), False), LlmProviders.LM_STUDIO: (lambda: litellm.LMStudioChatConfig(), False), + LlmProviders.FOUNDRY_LOCAL: ( + lambda: litellm.FoundryLocalChatConfig(), + False, + ), LlmProviders.GALADRIEL: (lambda: litellm.GaladrielChatConfig(), False), LlmProviders.REPLICATE: (lambda: litellm.ReplicateConfig(), False), LlmProviders.HUGGINGFACE: (lambda: litellm.HuggingFaceChatConfig(), False), diff --git a/provider_endpoints_support.json b/provider_endpoints_support.json index 388752b032e..a367b992e0f 100644 --- a/provider_endpoints_support.json +++ b/provider_endpoints_support.json @@ -1395,6 +1395,24 @@ "interactions": true } }, + "foundry_local": { + "display_name": "Foundry Local (`foundry_local`)", + "url": "https://docs.litellm.ai/docs/providers/foundry_local", + "endpoints": { + "chat_completions": true, + "messages": true, + "responses": true, + "embeddings": false, + "image_generations": false, + "audio_transcriptions": false, + "audio_speech": false, + "moderations": false, + "batches": false, + "rerank": false, + "a2a": true, + "interactions": true + } + }, "maritalk": { "display_name": "Maritalk (`maritalk`)", "url": "https://docs.litellm.ai/docs/providers/maritalk", diff --git a/tests/test_litellm/llms/foundry_local/test_foundry_local_chat_transformation.py b/tests/test_litellm/llms/foundry_local/test_foundry_local_chat_transformation.py new file mode 100644 index 00000000000..21aa3cfe49f --- /dev/null +++ b/tests/test_litellm/llms/foundry_local/test_foundry_local_chat_transformation.py @@ -0,0 +1,88 @@ +import os +import sys +from unittest.mock import patch + +sys.path.insert( + 0, os.path.abspath(os.path.join(os.path.dirname(__file__), "../../../../..")) +) + +import litellm +from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider +from litellm.llms.foundry_local.chat.transformation import FoundryLocalChatConfig + + +class TestFoundryLocalChatConfig: + def test_should_resolve_provider_from_model_prefix(self): + """get_llm_provider should return foundry_local for 'foundry_local/model' strings.""" + _, provider, _, _ = get_llm_provider("foundry_local/phi-3.5-mini") + assert provider == "foundry_local" + + def test_should_return_fake_api_key_when_none_provided(self): + """Foundry Local does not require auth; a fake key is returned for the OpenAI client.""" + config = FoundryLocalChatConfig() + _, api_key = config._get_openai_compatible_provider_info(None, None) + assert api_key == "fake-api-key" + + def test_should_use_explicit_api_key_when_provided(self): + config = FoundryLocalChatConfig() + _, api_key = config._get_openai_compatible_provider_info(None, "my-key") + assert api_key == "my-key" + + def test_should_use_explicit_api_base_when_provided(self): + config = FoundryLocalChatConfig() + api_base, _ = config._get_openai_compatible_provider_info( + "http://localhost:5272/v1", None + ) + assert api_base == "http://localhost:5272/v1" + + def test_should_read_env_vars_for_api_base_and_key(self): + config = FoundryLocalChatConfig() + with patch.dict( + "os.environ", + { + "FOUNDRY_LOCAL_API_BASE": "http://localhost:5272/v1", + "FOUNDRY_LOCAL_API_KEY": "env-key", + }, + ): + api_base, api_key = config._get_openai_compatible_provider_info(None, None) + assert api_base == "http://localhost:5272/v1" + assert api_key == "env-key" + + def test_should_prefer_explicit_over_env_vars(self): + config = FoundryLocalChatConfig() + with patch.dict( + "os.environ", + { + "FOUNDRY_LOCAL_API_BASE": "http://env-base:5272/v1", + "FOUNDRY_LOCAL_API_KEY": "env-key", + }, + ): + api_base, api_key = config._get_openai_compatible_provider_info( + "http://explicit:9999/v1", "explicit-key" + ) + assert api_base == "http://explicit:9999/v1" + assert api_key == "explicit-key" + + def test_should_be_in_openai_compatible_providers(self): + """foundry_local must be listed as an OpenAI-compatible provider.""" + assert "foundry_local" in litellm.openai_compatible_providers + + def test_should_be_in_provider_list(self): + """foundry_local must appear in the global provider list.""" + assert "foundry_local" in litellm.provider_list + + def test_should_resolve_provider_info_in_get_llm_provider(self): + """get_llm_provider should resolve api_base and api_key via env vars.""" + with patch.dict( + "os.environ", + { + "FOUNDRY_LOCAL_API_BASE": "http://localhost:5272/v1", + }, + ): + model, provider, dynamic_api_key, api_base = get_llm_provider( + "foundry_local/phi-3.5-mini" + ) + assert model == "phi-3.5-mini" + assert provider == "foundry_local" + assert api_base == "http://localhost:5272/v1" + assert dynamic_api_key == "fake-api-key" diff --git a/ui/litellm-dashboard/public/assets/logos/foundry_local.svg b/ui/litellm-dashboard/public/assets/logos/foundry_local.svg new file mode 100644 index 00000000000..412a6fb70d6 --- /dev/null +++ b/ui/litellm-dashboard/public/assets/logos/foundry_local.svg @@ -0,0 +1,40 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/ui/litellm-dashboard/src/components/provider_info_helpers.tsx b/ui/litellm-dashboard/src/components/provider_info_helpers.tsx index 105951114ca..f9196482b97 100644 --- a/ui/litellm-dashboard/src/components/provider_info_helpers.tsx +++ b/ui/litellm-dashboard/src/components/provider_info_helpers.tsx @@ -36,6 +36,7 @@ export enum Providers { ElevenLabs = "ElevenLabs", EMPOWER = "Empower", FalAI = "Fal AI", + FOUNDRY_LOCAL = "Foundry Local", FEATHERLESS_AI = "Featherless Ai", FireworksAI = "Fireworks AI", FRIENDLIAI = "Friendliai", @@ -143,6 +144,7 @@ export const provider_map: Record = { ElevenLabs: "elevenlabs", EMPOWER: "empower", FalAI: "fal_ai", + FOUNDRY_LOCAL: "foundry_local", FEATHERLESS_AI: "featherless_ai", FireworksAI: "fireworks_ai", FRIENDLIAI: "friendliai", @@ -246,6 +248,7 @@ export const providerLogoMap: Record = { [Providers.DeepInfra]: `${asset_logos_folder}deepinfra.png`, [Providers.ElevenLabs]: `${asset_logos_folder}elevenlabs.png`, [Providers.FalAI]: `${asset_logos_folder}fal_ai.jpg`, + [Providers.FOUNDRY_LOCAL]: `${asset_logos_folder}foundry_local.svg`, [Providers.FEATHERLESS_AI]: `${asset_logos_folder}featherless.svg`, [Providers.FireworksAI]: `${asset_logos_folder}fireworks.svg`, [Providers.FRIENDLIAI]: `${asset_logos_folder}friendli.svg`,