diff --git a/docs/my-website/docs/simple_proxy.md b/docs/my-website/docs/simple_proxy.md index a0889ba2a3f..0e7d5c35504 100644 --- a/docs/my-website/docs/simple_proxy.md +++ b/docs/my-website/docs/simple_proxy.md @@ -549,7 +549,7 @@ general_settings: master_key: sk-1234 # [OPTIONAL] Only use this if you to require all calls to contain this key (Authorization: Bearer sk-1234) ``` -### Quick Start - Config +### Multiple Models - Quick Start Here's how you can use multiple llms with one proxy `config.yaml`. @@ -601,6 +601,8 @@ curl --location 'http://0.0.0.0:8000/chat/completions' \ ``` + + ### Save Model-specific params (API Base, API Keys, Temperature, Headers etc.) You can use the config to save model-specific information like api_base, api_key, temperature, max_tokens, etc. @@ -654,6 +656,31 @@ model_list: model: ollama/llama2 ``` +### Multiple Instances of 1 model + +If you have multiple instances of the same model, + +in the `config.yaml` just add all of them with the same 'model_name', and the proxy will handle routing requests (using LiteLLM's Router). + +In the config below requests with `model=zephyr-beta` will be routed across multiple instances of `HuggingFaceH4/zephyr-7b-beta` + +```yaml +model_list: + - model_name: zephyr-beta + litellm_params: + model: huggingface/HuggingFaceH4/zephyr-7b-beta + api_base: http://0.0.0.0:8001 + - model_name: zephyr-beta + litellm_params: + model: huggingface/HuggingFaceH4/zephyr-7b-beta + api_base: http://0.0.0.0:8002 + - model_name: zephyr-beta + litellm_params: + model: huggingface/HuggingFaceH4/zephyr-7b-beta + api_base: http://0.0.0.0:8003 +``` + + ### Set Custom Prompt Templates LiteLLM by default checks if a model has a [prompt template and applies it](./completion/prompt_formatting.md) (e.g. if a huggingface model has a saved chat template in it's tokenizer_config.json). However, you can also set a custom prompt template on your proxy in the `config.yaml`: @@ -867,222 +894,3 @@ Expected output on Langfuse ``` - - - diff --git a/litellm/.env.template b/litellm/.env.template deleted file mode 100644 index 93495decbb9..00000000000 --- a/litellm/.env.template +++ /dev/null @@ -1,19 +0,0 @@ -### KEYS ### -# HUGGINGFACE_API_KEY="" # Uncomment to save your Hugging Face API key -# OPENAI_API_KEY="" # Uncomment to save your OpenAI API Key -# TOGETHERAI_API_KEY="" # Uncomment to save your TogetherAI API key -# NLP_CLOUD_API_KEY="" # Uncomment to save your NLP Cloud API key -# ANTHROPIC_API_KEY="" # Uncomment to save your Anthropic API key - -### MODEL CUSTOM PROMPT TEMPLATE ### -# MODEL_SYSTEM_MESSAGE_START_TOKEN = "<|prompter|>" # This does not need to be a token, can be any string -# MODEL_SYSTEM_MESSAGE_END_TOKEN = "<|endoftext|>" # This does not need to be a token, can be any string - -# MODEL_USER_MESSAGE_START_TOKEN = "<|prompter|>" # This does not need to be a token, can be any string -# MODEL_USER_MESSAGE_END_TOKEN = "<|endoftext|>" # Applies only to user messages. Can be any string. - -# MODEL_ASSISTANT_MESSAGE_START_TOKEN = "<|prompter|>" # Applies only to assistant messages. Can be any string. -# MODEL_ASSISTANT_MESSAGE_END_TOKEN = "<|endoftext|>" # Applies only to system messages. Can be any string. - -# MODEL_PRE_PROMPT = "You are a good bot" # Applied at the start of the prompt -# MODEL_POST_PROMPT = "Now answer as best as you can" # Applied at the end of the prompt \ No newline at end of file