From 5c550118881bf89e19f8b646893a644df9c550d3 Mon Sep 17 00:00:00 2001 From: ishaan-jaff Date: Wed, 4 Oct 2023 16:14:09 -0700 Subject: [PATCH] docs --- docs/my-website/docs/rate_limit_manager.md | 11 +++++++---- 1 file changed, 7 insertions(+), 4 deletions(-) diff --git a/docs/my-website/docs/rate_limit_manager.md b/docs/my-website/docs/rate_limit_manager.md index 6a412ec63b0..683be3d5b43 100644 --- a/docs/my-website/docs/rate_limit_manager.md +++ b/docs/my-website/docs/rate_limit_manager.md @@ -2,21 +2,23 @@ import Tabs from '@theme/Tabs'; import TabItem from '@theme/TabItem'; # Rate Limit Manager -`RateLimitManager` allows you to maximize throughput while staying under rate limits +`RateLimitManager` allows you to maximize throughput while staying under rate limits. You can use RateLimitManager to submit a batch of completion jobs to execute -## Quick start - using +## Quick start ```python import asyncio from litellm import RateLimitManager batch_completion +# init RateLimitManager with your rate limits handler = RateLimitManager( max_requests_per_minute = 60, max_tokens_per_minute = 20000 ) -##### USAGE ################ +# make batch completion requests +# define jobs, list of kwargs to go to the completion() call jobs = [ {"model": "gpt-3.5-turbo-16k", "messages": [{"content": "Please provide a summary of the latest scientific discoveries.", "role": "user"}]}, {"model": "gpt-3.5-turbo-16k", "messages": [{"content": "Please provide a summary of the latest scientific discoveries.", "role": "user"}]}, @@ -25,7 +27,8 @@ jobs = [ {"model": "gpt-3.5-turbo-16k", "messages": [{"content": "Please provide a summary of the latest scientific discoveries.", "role": "user"}]} ] - +# use RateLimitManager.batch_completion to execute several jobs in parallel +# output stored in litellm_results.jsonl try: asyncio.run( handler.batch_completion(