From 2710bec02db713893219a2a224e32313e2d612b6 Mon Sep 17 00:00:00 2001 From: Krrish Dholakia Date: Wed, 7 Aug 2024 21:49:06 -0700 Subject: [PATCH] docs(scheduler.md): cleanup docs to use /chat/completion endpoint --- docs/my-website/docs/scheduler.md | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/docs/my-website/docs/scheduler.md b/docs/my-website/docs/scheduler.md index 8329fc2adbe..e59b03eacb7 100644 --- a/docs/my-website/docs/scheduler.md +++ b/docs/my-website/docs/scheduler.md @@ -41,7 +41,7 @@ router = Router( ) try: - _response = await router.schedule_acompletion( # 👈 ADDS TO QUEUE + POLLS + MAKES CALL + _response = await router.acompletion( # 👈 ADDS TO QUEUE + POLLS + MAKES CALL model="gpt-3.5-turbo", messages=[{"role": "user", "content": "Hey!"}], priority=0, # 👈 LOWER IS BETTER @@ -52,13 +52,13 @@ except Exception as e: ## LiteLLM Proxy -To prioritize requests on LiteLLM Proxy call our beta openai-compatible `http://localhost:4000/queue` endpoint. +To prioritize requests on LiteLLM Proxy add `priority` to the request. ```curl -curl -X POST 'http://localhost:4000/queue/chat/completions' \ +curl -X POST 'http://localhost:4000/chat/completions' \ -H 'Content-Type: application/json' \ -H 'Authorization: Bearer sk-1234' \ -D '{ @@ -128,7 +128,7 @@ router = Router( ) try: - _response = await router.schedule_acompletion( # 👈 ADDS TO QUEUE + POLLS + MAKES CALL + _response = await router.acompletion( # 👈 ADDS TO QUEUE + POLLS + MAKES CALL model="gpt-3.5-turbo", messages=[{"role": "user", "content": "Hey!"}], priority=0, # 👈 LOWER IS BETTER