diff --git a/docs/my-website/docs/completion/stream.md b/docs/my-website/docs/completion/stream.md index da578f7dee8..6cb813ae934 100644 --- a/docs/my-website/docs/completion/stream.md +++ b/docs/my-website/docs/completion/stream.md @@ -34,9 +34,11 @@ print(response) ## Async Streaming We've implemented an `__anext__()` function in the streaming object returned. This enables async iteration over the streaming object. + ### Usage +Here's an example of using it with openai. But this ``` -from litellm import acompletion +from litellm import completion import asyncio def logger_fn(model_call_object: dict): @@ -46,11 +48,10 @@ def logger_fn(model_call_object: dict): user_message = "Hello, how are you?" messages = [{"content": user_message, "role": "user"}] -# # test on ai21 completion call -async def ai21_async_completion_call(): +async def completion_call(): try: response = completion( - model="j2-ultra", messages=messages, stream=True, logger_fn=logger_fn + model="gpt-3.5-turbo", messages=messages, stream=True, logger_fn=logger_fn ) print(f"response: {response}") complete_response = "" @@ -67,5 +68,5 @@ async def ai21_async_completion_call(): print(f"error occurred: {traceback.format_exc()}") pass -asyncio.run(ai21_async_completion_call()) +asyncio.run(completion_call()) ``` \ No newline at end of file diff --git a/litellm/__pycache__/__init__.cpython-311.pyc b/litellm/__pycache__/__init__.cpython-311.pyc index a95ce65c852..6c515ddae08 100644 Binary files a/litellm/__pycache__/__init__.cpython-311.pyc and b/litellm/__pycache__/__init__.cpython-311.pyc differ diff --git a/litellm/__pycache__/main.cpython-311.pyc b/litellm/__pycache__/main.cpython-311.pyc index d7f053f664c..adb45149c93 100644 Binary files a/litellm/__pycache__/main.cpython-311.pyc and b/litellm/__pycache__/main.cpython-311.pyc differ diff --git a/litellm/integrations/__pycache__/supabase.cpython-311.pyc b/litellm/integrations/__pycache__/supabase.cpython-311.pyc index ffb30cda564..ffc824e92a5 100644 Binary files a/litellm/integrations/__pycache__/supabase.cpython-311.pyc and b/litellm/integrations/__pycache__/supabase.cpython-311.pyc differ diff --git a/litellm/tests/test_streaming.py b/litellm/tests/test_streaming.py index e5b9c3993ef..d37cc1c4ef8 100644 --- a/litellm/tests/test_streaming.py +++ b/litellm/tests/test_streaming.py @@ -1,7 +1,7 @@ #### What this tests #### # This tests streaming for the completion endpoint -import sys, os +import sys, os, asyncio import traceback import time @@ -102,6 +102,27 @@ def test_openai_chat_completion_call(): print(f"error occurred: {traceback.format_exc()}") pass +async def completion_call(): + try: + response = completion( + model="gpt-3.5-turbo", messages=messages, stream=True, logger_fn=logger_fn + ) + print(f"response: {response}") + complete_response = "" + start_time = time.time() + # Change for loop to async for loop + async for chunk in response: + chunk_time = time.time() + print(f"time since initial request: {chunk_time - start_time:.5f}") + print(chunk["choices"][0]["delta"]) + complete_response += chunk["choices"][0]["delta"]["content"] + if complete_response == "": + raise Exception("Empty response received") + except: + print(f"error occurred: {traceback.format_exc()}") + pass + +asyncio.run(completion_call()) # # test on azure completion call # try: @@ -191,7 +212,7 @@ def test_together_ai_completion_call_starcoder(): print(f"error occurred: {traceback.format_exc()}") pass -# test on aleph alpha completion call +# test on aleph alpha completion call - commented out as it's expensive to run this on circle ci for every build # def test_aleph_alpha_call(): # try: # start_time = time.time()