From 61d2e91632cdb2c1b4a59419e003395356c5b6dc Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 23 Mar 2024 17:39:07 -0700 Subject: [PATCH 1/2] (docs) update gunicorn usage --- docs/my-website/docs/proxy/deploy.md | 19 ++++--------------- 1 file changed, 4 insertions(+), 15 deletions(-) diff --git a/docs/my-website/docs/proxy/deploy.md b/docs/my-website/docs/proxy/deploy.md index d25035760c7..acd61e625bd 100644 --- a/docs/my-website/docs/proxy/deploy.md +++ b/docs/my-website/docs/proxy/deploy.md @@ -103,7 +103,10 @@ RUN chmod +x entrypoint.sh EXPOSE 4000/tcp # Override the CMD instruction with your desired command and arguments -CMD ["--port", "4000", "--config", "config.yaml", "--detailed_debug", "--run_gunicorn"] +# WARNING: FOR PROD DO NOT USE `--detailed_debug` it slows down response times, instead use the following CMD +# CMD ["--port", "4000", "--config", "config.yaml"] + +CMD ["--port", "4000", "--config", "config.yaml", "--detailed_debug"] ``` @@ -478,20 +481,6 @@ ghcr.io/berriai/litellm-database:main-latest --config your_config.yaml ### 1. Switch of debug logs in production don't use [`--detailed-debug`, `--debug`](https://docs.litellm.ai/docs/proxy/debugging#detailed-debug) or `litellm.set_verbose=True`. We found using debug logs can add 5-10% latency per LLM API call -### 2. Use `run_gunicorn` and `num_workers` - -Example setting `--run_gunicorn` and `--num_workers` -```shell -docker run ghcr.io/berriai/litellm-database:main-latest --run_gunicorn --num_workers 4 -``` - -Why `Gunicorn`? -- Gunicorn takes care of running multiple instances of your web application -- Gunicorn is ideal for running litellm proxy on cluster of machines with Kubernetes - -Why `num_workers`? -Setting `num_workers` to the number of CPUs available ensures optimal utilization of system resources by matching the number of worker processes to the available CPU cores. - ## Advanced Deployment Settings From 19a1d999ec528d673f1dc5a1fdac17823a133c44 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 23 Mar 2024 17:40:22 -0700 Subject: [PATCH 2/2] (feat) update docs to not include gunicorn usage --- Dockerfile | 4 ++-- Dockerfile.database | 4 ++-- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/Dockerfile b/Dockerfile index 71d96ff2012..3262d42b5fa 100644 --- a/Dockerfile +++ b/Dockerfile @@ -70,5 +70,5 @@ EXPOSE 4000/tcp ENTRYPOINT ["litellm"] # Append "--detailed_debug" to the end of CMD to view detailed debug logs -# CMD ["--port", "4000", "--config", "./proxy_server_config.yaml", "--run_gunicorn", "--detailed_debug"] -CMD ["--port", "4000", "--config", "./proxy_server_config.yaml", "--run_gunicorn", "--num_workers", "4"] +# CMD ["--port", "4000", "--config", "./proxy_server_config.yaml"] +CMD ["--port", "4000", "--config", "./proxy_server_config.yaml"] diff --git a/Dockerfile.database b/Dockerfile.database index 57505c32dc2..22084bab89c 100644 --- a/Dockerfile.database +++ b/Dockerfile.database @@ -72,5 +72,5 @@ EXPOSE 4000/tcp ENTRYPOINT ["litellm"] # Append "--detailed_debug" to the end of CMD to view detailed debug logs -# CMD ["--port", "4000","--run_gunicorn", "--detailed_debug"] -CMD ["--port", "4000", "--run_gunicorn"] +# CMD ["--port", "4000", "--detailed_debug"] +CMD ["--port", "4000"]