diff --git a/docs/my-website/docs/proxy/load_balancing.md b/docs/my-website/docs/proxy/load_balancing.md index 0e1f58662cb..a73cfe33211 100644 --- a/docs/my-website/docs/proxy/load_balancing.md +++ b/docs/my-website/docs/proxy/load_balancing.md @@ -3,38 +3,39 @@ Load balance multiple instances of the same model The proxy will handle routing requests (using LiteLLM's Router). **Set `rpm` in the config if you want maximize throughput** +## Quick Start - Load Balancing +### Step 1 - Set deployments on config -#### Example config -requests with `model=gpt-3.5-turbo` will be routed across multiple instances of `azure/gpt-3.5-turbo` +**Example config below**. Here requests with `model=gpt-3.5-turbo` will be routed across multiple instances of `azure/gpt-3.5-turbo` ```yaml model_list: - model_name: gpt-3.5-turbo litellm_params: - model: azure/gpt-turbo-small-eu - api_base: https://my-endpoint-europe-berri-992.openai.azure.com/ - api_key: + model: azure/ + api_base: + api_key: rpm: 6 # Rate limit for this deployment: in requests per minute (rpm) - model_name: gpt-3.5-turbo litellm_params: model: azure/gpt-turbo-small-ca api_base: https://my-endpoint-canada-berri992.openai.azure.com/ - api_key: + api_key: rpm: 6 - model_name: gpt-3.5-turbo litellm_params: model: azure/gpt-turbo-large api_base: https://openai-france-1234.openai.azure.com/ - api_key: + api_key: rpm: 1440 ``` -#### Step 2: Start Proxy with config +### Step 2: Start Proxy with config ```shell $ litellm --config /path/to/config.yaml ``` -#### Step 3: Use proxy +### Step 3: Use proxy - Call a model group [Load Balancing] Curl Command ```shell curl --location 'http://0.0.0.0:8000/chat/completions' \ @@ -51,7 +52,28 @@ curl --location 'http://0.0.0.0:8000/chat/completions' \ ' ``` -### Fallbacks + Cooldowns + Retries + Timeouts +### Usage - Call a specific model deployment +If you want to call a specific model defined in the `config.yaml`, you can call the `litellm_params: model` + +In this example it will call `azure/gpt-turbo-small-ca`. Defined in the config on Step 1 + +```bash +curl --location 'http://0.0.0.0:8000/chat/completions' \ +--header 'Content-Type: application/json' \ +--data ' { + "model": "azure/gpt-turbo-small-ca", + "messages": [ + { + "role": "user", + "content": "what llm are you" + } + ], + } +' +``` + + +## Fallbacks + Cooldowns + Retries + Timeouts If a call fails after num_retries, fall back to another model group. @@ -122,7 +144,7 @@ model_list: api_base: https://my-endpoint-europe-berri-992.openai.azure.com/ api_key: timeout: 0.1 # timeout in (seconds) - stream_timeout: 0.01 # timeout stream requests (seconds) + stream_timeout: 0.01 # timeout for stream requests (seconds) max_retries: 5 - model_name: gpt-3.5-turbo litellm_params: @@ -130,7 +152,7 @@ model_list: api_base: https://my-endpoint-canada-berri992.openai.azure.com/ api_key: timeout: 0.1 # timeout in (seconds) - stream_timeout: 0.01 # timeout stream requests (seconds) + stream_timeout: 0.01 # timeout for stream requests (seconds) max_retries: 5 ```