diff --git a/.gitmodules b/.gitmodules new file mode 100644 index 00000000000..2727919abf5 --- /dev/null +++ b/.gitmodules @@ -0,0 +1,3 @@ +[submodule "fastrepl-proxy"] + path = fastrepl-proxy + url = https://github.com/BerriAI/litellm diff --git a/docs/my-website/docs/completion/mock_requests.md b/docs/my-website/docs/completion/mock_requests.md index dd9b4a4f79d..fc357b0d7d7 100644 --- a/docs/my-website/docs/completion/mock_requests.md +++ b/docs/my-website/docs/completion/mock_requests.md @@ -1,4 +1,4 @@ -# Mock Requests - Save Testing Costs 💰 +# Mock Completion() Responses - Save Testing Costs 💰 For testing purposes, you can use `completion()` with `mock_response` to mock calling the completion endpoint. diff --git a/docs/my-website/docs/max_tokens_cost.md b/docs/my-website/docs/max_tokens_cost.md new file mode 100644 index 00000000000..5447f2e66a6 --- /dev/null +++ b/docs/my-website/docs/max_tokens_cost.md @@ -0,0 +1,34 @@ +# get context window & cost per token + +100+ LLMs supported: See full list [here](https://github.com/BerriAI/litellm/blob/main/model_prices_and_context_window.json) + +## using api.litellm.ai +```shell +curl 'https://api.litellm.ai/get_max_tokens?model=claude-2' +``` + +### output +```json +{ + "input_cost_per_token": 1.102e-05, + "max_tokens": 100000, + "model": "claude-2", + "output_cost_per_token": 3.268e-05 +} +``` + +## using the litellm python package +```python +import litellm +model_data = litellm.model_cost["gpt-4"] +``` + +### output +```json +{ + "input_cost_per_token": 3e-06, + "max_tokens": 8192, + "model": "gpt-4", + "output_cost_per_token": 6e-05 +} +``` \ No newline at end of file diff --git a/docs/my-website/docs/providers/bedrock.md b/docs/my-website/docs/providers/bedrock.md index 08b72a484dd..9f86f67d386 100644 --- a/docs/my-website/docs/providers/bedrock.md +++ b/docs/my-website/docs/providers/bedrock.md @@ -1,13 +1,13 @@ # AWS Bedrock -### API KEYS +## API KEYS ```python os.environ["AWS_ACCESS_KEY_ID"] = "" os.environ["AWS_SECRET_ACCESS_KEY"] = "" os.environ["AWS_REGION_NAME"] = "" ``` -### Usage +## Usage ```python import os from litellm import completion @@ -24,7 +24,7 @@ response = completion( ) ``` -### Supported AWS Bedrock Models +## Supported AWS Bedrock Models Here's an example of using a bedrock model with LiteLLM | Model Name | Function Call | Required OS Variables | @@ -33,7 +33,54 @@ Here's an example of using a bedrock model with LiteLLM | AI21 J2-Ultra | `completion(model='bedrock/ai21.j2-ultra', messages=messages)` | `os.environ['AWS_ACCESS_KEY_ID']`, `os.environ['AWS_SECRET_ACCESS_KEY']`, `os.environ['AWS_REGION_NAME']` | | AI21 J2-Mid | `completion(model='bedrock/ai21.j2-mid', messages=messages)` | `os.environ['AWS_ACCESS_KEY_ID']`, `os.environ['AWS_SECRET_ACCESS_KEY']`, `os.environ['AWS_REGION_NAME']` | -### Troubleshooting +## Streaming +Bedrock currently supports streaming for the following llms: +* `bedrock/amazon.titan-tg1-large` + +Example Usage +```python +import os +from litellm import completion + +os.environ["AWS_ACCESS_KEY_ID"] = "" +os.environ["AWS_SECRET_ACCESS_KEY"] = "" +os.environ["AWS_REGION_NAME"] = "" + +response = completion( + model="bedrock/amazon.titan-tg1-large", + messages=[{ "content": "Hello, how are you?","role": "user"}], + temperature=0.2, + max_tokens=80, + stream=True +) + +for chunk in response: + print(chunk) +``` + +### Example Streaming Output Chunk +```json +{ + "choices": [ + { + "finish_reason": null, + "index": 0, + "delta": { + "content": "ase can appeal the case to a higher federal court. If a higher federal court rules in a way that conflicts with a ruling from a lower federal court or conflicts with a ruling from a higher state court, the parties involved in the case can appeal the case to the Supreme Court. In order to appeal a case to the Sup" + } + } + ], + "created": null, + "model": "amazon.titan-tg1-large", + "usage": { + "prompt_tokens": null, + "completion_tokens": null, + "total_tokens": null + } +} +``` + +## Troubleshooting If creating a boto3 bedrock client fails with `Unknown service: 'bedrock'` Try re installing boto3 using the following commands ```shell diff --git a/docs/my-website/sidebars.js b/docs/my-website/sidebars.js index b30e84aff76..d875a097b0a 100644 --- a/docs/my-website/sidebars.js +++ b/docs/my-website/sidebars.js @@ -79,6 +79,7 @@ const sidebars = { "token_usage", "exception_mapping", 'debugging/local_debugging', + "max_tokens_cost", "budget_manager", "proxy_api", { diff --git a/fastrepl-proxy b/fastrepl-proxy new file mode 160000 index 00000000000..f2fe83e002a --- /dev/null +++ b/fastrepl-proxy @@ -0,0 +1 @@ +Subproject commit f2fe83e002a7c3ddedf4e500665644adfd31b9fc diff --git a/litellm/__pycache__/__init__.cpython-311.pyc b/litellm/__pycache__/__init__.cpython-311.pyc index 4dd38f2b1b1..d5e525906ab 100644 Binary files a/litellm/__pycache__/__init__.cpython-311.pyc and b/litellm/__pycache__/__init__.cpython-311.pyc differ diff --git a/litellm/__pycache__/main.cpython-311.pyc b/litellm/__pycache__/main.cpython-311.pyc index 48f5427417d..5e9be16631e 100644 Binary files a/litellm/__pycache__/main.cpython-311.pyc and b/litellm/__pycache__/main.cpython-311.pyc differ diff --git a/litellm/__pycache__/utils.cpython-311.pyc b/litellm/__pycache__/utils.cpython-311.pyc index 7e2dcf510a0..61ed4069b3e 100644 Binary files a/litellm/__pycache__/utils.cpython-311.pyc and b/litellm/__pycache__/utils.cpython-311.pyc differ diff --git a/litellm/llms/bedrock.py b/litellm/llms/bedrock.py index f7b39acea10..7c885a18728 100644 --- a/litellm/llms/bedrock.py +++ b/litellm/llms/bedrock.py @@ -59,6 +59,7 @@ def completion( encoding, logging_obj, optional_params=None, + stream=False, litellm_params=None, logger_fn=None, ): @@ -94,14 +95,8 @@ def completion( else: # amazon titan data = json.dumps({ "inputText": prompt, - "textGenerationConfig":{ - "maxTokenCount":4096, - "stopSequences":[], - "temperature":0, - "topP":0.9 - } + "textGenerationConfig": optional_params, }) - ## LOGGING logging_obj.pre_call( input=prompt, @@ -112,6 +107,15 @@ def completion( ## COMPLETION CALL accept = 'application/json' contentType = 'application/json' + if stream == True: + response = client.invoke_model_with_response_stream( + body=data, + modelId=model, + accept=accept, + contentType=contentType + ) + response = response.get('body') + return response response = client.invoke_model( body=data, @@ -120,50 +124,48 @@ def completion( contentType=contentType ) response_body = json.loads(response.get('body').read()) - if "stream" in optional_params and optional_params["stream"] == True: - return response.iter_lines() - else: - ## LOGGING - logging_obj.post_call( - input=prompt, - api_key="", - original_response=response, - additional_args={"complete_input_dict": data}, - ) - print_verbose(f"raw model_response: {response}") - ## RESPONSE OBJECT - outputText = "default" - if provider == "ai21": - outputText = response_body.get('completions')[0].get('data').get('text') - else: # amazon titan - outputText = response_body.get('results')[0].get('outputText') - if "error" in outputText: - raise BedrockError( - message=outputText, - status_code=response.status_code, - ) - else: - try: - model_response["choices"][0]["message"]["content"] = outputText - except: - raise BedrockError(message=json.dumps(outputText), status_code=response.status_code) - ## CALCULATING USAGE - baseten charges on time, not tokens - have some mapping of cost here. - prompt_tokens = len( - encoding.encode(prompt) - ) - completion_tokens = len( - encoding.encode(model_response["choices"][0]["message"]["content"]) + ## LOGGING + logging_obj.post_call( + input=prompt, + api_key="", + original_response=response, + additional_args={"complete_input_dict": data}, ) + print_verbose(f"raw model_response: {response}") + ## RESPONSE OBJECT + outputText = "default" + if provider == "ai21": + outputText = response_body.get('completions')[0].get('data').get('text') + else: # amazon titan + outputText = response_body.get('results')[0].get('outputText') + if "error" in outputText: + raise BedrockError( + message=outputText, + status_code=response.status_code, + ) + else: + try: + model_response["choices"][0]["message"]["content"] = outputText + except: + raise BedrockError(message=json.dumps(outputText), status_code=response.status_code) - model_response["created"] = time.time() - model_response["model"] = model - model_response["usage"] = { - "prompt_tokens": prompt_tokens, - "completion_tokens": completion_tokens, - "total_tokens": prompt_tokens + completion_tokens, - } - return model_response + ## CALCULATING USAGE - baseten charges on time, not tokens - have some mapping of cost here. + prompt_tokens = len( + encoding.encode(prompt) + ) + completion_tokens = len( + encoding.encode(model_response["choices"][0]["message"]["content"]) + ) + + model_response["created"] = time.time() + model_response["model"] = model + model_response["usage"] = { + "prompt_tokens": prompt_tokens, + "completion_tokens": completion_tokens, + "total_tokens": prompt_tokens + completion_tokens, + } + return model_response def embedding(): # logic for parsing in - calling - parsing out model embedding calls diff --git a/litellm/main.py b/litellm/main.py index 518da5afaf9..23cfb2c5819 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -82,8 +82,7 @@ def mock_completion(model: str, messages: List, stream: bool = False, mock_respo response = mock_completion_streaming_obj(model_response, mock_response=mock_response, model=model) return response - completion_response = "This is a mock request" - model_response["choices"][0]["message"]["content"] = completion_response + model_response["choices"][0]["message"]["content"] = mock_response model_response["created"] = time.time() model_response["model"] = model return model_response @@ -133,6 +132,7 @@ def completion( # model specific optional params top_k=40,# used by text-bison only task: Optional[str]="text-generation-inference", # used by huggingface inference endpoints + return_full_text: bool = False, # used by huggingface TGI remove_input: bool = True, # used by nlp cloud models - prevents input text from being returned as part of output request_timeout=0, # unused var for old version of OpenAI API fallbacks=[], @@ -162,6 +162,7 @@ def completion( ): # allow custom provider to be passed in via the model name "azure/chatgpt-test" custom_llm_provider = model.split("/", 1)[0] model = model.split("/", 1)[1] + model, custom_llm_provider = get_llm_provider(model=model, custom_llm_provider=custom_llm_provider) # check if user passed in any of the OpenAI optional params optional_params = get_optional_params( functions=functions, @@ -182,7 +183,8 @@ def completion( custom_llm_provider=custom_llm_provider, top_k=top_k, task=task, - remove_input=remove_input + remove_input=remove_input, + return_full_text=return_full_text ) # For logging - save the values of the litellm-specific params passed in litellm_params = get_litellm_params( @@ -198,7 +200,6 @@ def completion( completion_call_id=id ) logging.update_environment_variables(model=model, user=user, optional_params=optional_params, litellm_params=litellm_params) - get_llm_provider(model=model, custom_llm_provider=custom_llm_provider) if custom_llm_provider == "azure": # azure configs api_type = get_secret("AZURE_API_TYPE") or "azure" @@ -244,7 +245,7 @@ def completion( **optional_params, ) if "stream" in optional_params and optional_params["stream"] == True: - response = CustomStreamWrapper(response, model, logging_obj=logging) + response = CustomStreamWrapper(response, model, custom_llm_provider="openai", logging_obj=logging) return response ## LOGGING logging.post_call( @@ -280,7 +281,6 @@ def completion( litellm.openai_key or get_secret("OPENAI_API_KEY") ) - ## LOGGING logging.pre_call( input=messages, @@ -310,7 +310,7 @@ def completion( raise e if "stream" in optional_params and optional_params["stream"] == True: - response = CustomStreamWrapper(response, model, logging_obj=logging) + response = CustomStreamWrapper(response, model, custom_llm_provider="openai", logging_obj=logging) return response ## LOGGING logging.post_call( @@ -374,7 +374,7 @@ def completion( **optional_params ) if "stream" in optional_params and optional_params["stream"] == True: - response = CustomStreamWrapper(response, model, logging_obj=logging) + response = CustomStreamWrapper(response, model, custom_llm_provider="text-completion-openai", logging_obj=logging) return response ## LOGGING logging.post_call( @@ -446,7 +446,7 @@ def completion( ) if "stream" in optional_params and optional_params["stream"] == True: # don't try to access stream object, - response = CustomStreamWrapper(model_response, model, logging_obj=logging) + response = CustomStreamWrapper(model_response, model, custom_llm_provider="anthropic", logging_obj=logging) return response response = model_response elif model in litellm.nlp_cloud_models or custom_llm_provider == "nlp_cloud": @@ -493,7 +493,7 @@ def completion( if "stream" in optional_params and optional_params["stream"] == True: # don't try to access stream object, - response = CustomStreamWrapper(model_response, model, logging_obj=logging) + response = CustomStreamWrapper(model_response, model, custom_llm_provider="aleph-alpha", logging_obj=logging) return response response = model_response elif model in litellm.openrouter_models or custom_llm_provider == "openrouter": @@ -570,7 +570,7 @@ def completion( if "stream" in optional_params and optional_params["stream"] == True: # don't try to access stream object, - response = CustomStreamWrapper(model_response, model, logging_obj=logging) + response = CustomStreamWrapper(model_response, model, custom_llm_provider="cohere", logging_obj=logging) return response response = model_response elif ( @@ -782,10 +782,12 @@ def completion( litellm_params=litellm_params, logger_fn=logger_fn, encoding=encoding, - logging_obj=logging + logging_obj=logging, + stream=stream, ) - if "stream" in optional_params and optional_params["stream"] == True: ## [BETA] + + if stream == True: # don't try to access stream object, response = CustomStreamWrapper( iter(model_response), model, custom_llm_provider="bedrock", logging_obj=logging diff --git a/litellm/tests/__pycache__/test_bad_params.cpython-311-pytest-7.4.0.pyc b/litellm/tests/__pycache__/test_bad_params.cpython-311-pytest-7.4.0.pyc deleted file mode 100644 index 753da9901bc..00000000000 Binary files a/litellm/tests/__pycache__/test_bad_params.cpython-311-pytest-7.4.0.pyc and /dev/null differ diff --git a/litellm/tests/__pycache__/test_client.cpython-311-pytest-7.4.0.pyc b/litellm/tests/__pycache__/test_client.cpython-311-pytest-7.4.0.pyc deleted file mode 100644 index 33522b54aa3..00000000000 Binary files a/litellm/tests/__pycache__/test_client.cpython-311-pytest-7.4.0.pyc and /dev/null differ diff --git a/litellm/tests/__pycache__/test_completion.cpython-311-pytest-7.4.0.pyc b/litellm/tests/__pycache__/test_completion.cpython-311-pytest-7.4.0.pyc deleted file mode 100644 index 0c5fa166d7c..00000000000 Binary files a/litellm/tests/__pycache__/test_completion.cpython-311-pytest-7.4.0.pyc and /dev/null differ diff --git a/litellm/tests/__pycache__/test_exceptions.cpython-311-pytest-7.4.0.pyc b/litellm/tests/__pycache__/test_exceptions.cpython-311-pytest-7.4.0.pyc deleted file mode 100644 index 62f9422f80e..00000000000 Binary files a/litellm/tests/__pycache__/test_exceptions.cpython-311-pytest-7.4.0.pyc and /dev/null differ diff --git a/litellm/tests/__pycache__/test_logging.cpython-311-pytest-7.4.0.pyc b/litellm/tests/__pycache__/test_logging.cpython-311-pytest-7.4.0.pyc deleted file mode 100644 index 9f71ef3a112..00000000000 Binary files a/litellm/tests/__pycache__/test_logging.cpython-311-pytest-7.4.0.pyc and /dev/null differ diff --git a/litellm/tests/__pycache__/test_model_fallback.cpython-311-pytest-7.4.0.pyc b/litellm/tests/__pycache__/test_model_fallback.cpython-311-pytest-7.4.0.pyc deleted file mode 100644 index 864247d098e..00000000000 Binary files a/litellm/tests/__pycache__/test_model_fallback.cpython-311-pytest-7.4.0.pyc and /dev/null differ diff --git a/litellm/tests/__pycache__/test_timeout.cpython-311-pytest-7.4.0.pyc b/litellm/tests/__pycache__/test_timeout.cpython-311-pytest-7.4.0.pyc deleted file mode 100644 index f291f8a8d03..00000000000 Binary files a/litellm/tests/__pycache__/test_timeout.cpython-311-pytest-7.4.0.pyc and /dev/null differ diff --git a/litellm/tests/test_completion.py b/litellm/tests/test_completion.py index ab69ecfd4e7..9fadd2fcac4 100644 --- a/litellm/tests/test_completion.py +++ b/litellm/tests/test_completion.py @@ -91,60 +91,22 @@ def test_completion_with_litellm_call_id(): except Exception as e: pytest.fail(f"Error occurred: {e}") +# commenting out as this is a flaky test on circle ci +# def test_completion_nlp_cloud(): +# try: +# messages = [ +# {"role": "system", "content": "You are a helpful assistant."}, +# { +# "role": "user", +# "content": "how does a court case get to the Supreme Court?", +# }, +# ] +# response = completion(model="dolphin", messages=messages, logger_fn=logger_fn) +# print(response) +# except Exception as e: +# pytest.fail(f"Error occurred: {e}") -def test_completion_claude_stream(): - try: - messages = [ - {"role": "system", "content": "You are a helpful assistant."}, - { - "role": "user", - "content": "how does a court case get to the Supreme Court?", - }, - ] - response = completion(model="claude-2", messages=messages, stream=True) - # Add any assertions here to check the response - for chunk in response: - print(chunk["choices"][0]["delta"]) # same as openai format - print(chunk["choices"][0]["finish_reason"]) - print(chunk["choices"][0]["delta"]["content"]) - except Exception as e: - pytest.fail(f"Error occurred: {e}") -# test_completion_claude_stream() - -def test_completion_nlp_cloud(): - try: - messages = [ - {"role": "system", "content": "You are a helpful assistant."}, - { - "role": "user", - "content": "how does a court case get to the Supreme Court?", - }, - ] - response = completion(model="dolphin", messages=messages, logger_fn=logger_fn) - print(response) - except Exception as e: - pytest.fail(f"Error occurred: {e}") - -def test_completion_nlp_cloud_streaming(): - try: - messages = [ - {"role": "system", "content": "You are a helpful assistant."}, - { - "role": "user", - "content": "how does a court case get to the Supreme Court?", - }, - ] - response = completion(model="dolphin", messages=messages, stream=True, logger_fn=logger_fn) - # Add any assertions here to check the response - for chunk in response: - print(chunk["choices"][0]["delta"]["content"]) # same as openai format - print(chunk["choices"][0]["finish_reason"]) - print(chunk["choices"][0]["delta"]["content"]) - except Exception as e: - pytest.fail(f"Error occurred: {e}") -# test_completion_nlp_cloud_streaming() - -# test_completion_nlp_cloud_streaming() +# test_completion_nlp_cloud() # def test_completion_hf_api(): # try: # user_message = "write some code to find the sum of two numbers" @@ -193,29 +155,6 @@ def test_completion_cohere(): # commenting for now as the cohere endpoint is bei # test_completion_cohere() -def test_completion_cohere_stream(): - try: - messages = [ - {"role": "system", "content": "You are a helpful assistant."}, - { - "role": "user", - "content": "how does a court case get to the Supreme Court?", - }, - ] - response = completion( - model="command-nightly", messages=messages, stream=True, max_tokens=50 - ) - # Add any assertions here to check the response - for chunk in response: - print(chunk["choices"][0]["delta"]) # same as openai format - print(chunk["choices"][0]["finish_reason"]) - print(chunk["choices"][0]["delta"]["content"]) - except KeyError: - pass - except Exception as e: - pytest.fail(f"Error occurred: {e}") -# test_completion_cohere_stream() - def test_completion_openai(): try: @@ -350,70 +289,6 @@ def test_completion_openai_with_more_optional_params(): pytest.fail(f"Error occurred: {e}") -def test_completion_openai_with_stream(): - try: - response = completion( - model="gpt-3.5-turbo", - messages=messages, - temperature=0.5, - top_p=0.1, - n=2, - max_tokens=150, - presence_penalty=0.5, - stream=True, - frequency_penalty=-0.5, - logit_bias={27000: 5}, - user="ishaan_dev@berri.ai", - ) - # Add any assertions here to check the response - print(response) - for chunk in response: - print(chunk) - if chunk["choices"][0]["finish_reason"] == "stop" or chunk["choices"][0]["finish_reason"] == "length": - break - print(chunk["choices"][0]["finish_reason"]) - print(chunk["choices"][0]["delta"]["content"]) - except Exception as e: - pytest.fail(f"Error occurred: {e}") -# test_completion_openai_with_stream() - - -def test_completion_openai_with_functions(): - function1 = [ - { - "name": "get_current_weather", - "description": "Get the current weather in a given location", - "parameters": { - "type": "object", - "properties": { - "location": { - "type": "string", - "description": "The city and state, e.g. San Francisco, CA", - }, - "unit": {"type": "string", "enum": ["celsius", "fahrenheit"]}, - }, - "required": ["location"], - }, - } - ] - try: - response = completion( - model="gpt-3.5-turbo", messages=messages, functions=function1, stream=True - ) - # Add any assertions here to check the response - print(response) - for chunk in response: - print(chunk) - if chunk["choices"][0]["finish_reason"] == "stop": - break - print(chunk["choices"][0]["finish_reason"]) - print(chunk["choices"][0]["delta"]["content"]) - - except Exception as e: - pytest.fail(f"Error occurred: {e}") -# test_completion_openai_with_functions() - - # def test_completion_openai_azure_with_functions(): # function1 = [ # { @@ -568,20 +443,6 @@ def test_completion_replicate_vicuna(): except Exception as e: pytest.fail(f"Error occurred: {e}") -# test_completion_replicate_vicuna() - -def test_completion_replicate_llama_stream(): - model_name = "replicate/llama-2-70b-chat:2c1608e18606fad2812020dc541930f2d0495ce32eee50074220b87300bc16e1" - try: - response = completion(model=model_name, messages=messages, stream=True) - # Add any assertions here to check the response - for chunk in response: - print(chunk) - print(chunk["choices"][0]["delta"]["content"]) - except Exception as e: - pytest.fail(f"Error occurred: {e}") -# test_completion_replicate_llama_stream() - # def test_completion_replicate_stability_stream(): # model_name = "stability-ai/stablelm-tuned-alpha-7b:c49dae362cbaecd2ceabb5bd34fdb68413c4ff775111fea065d259d577757beb" # try: @@ -615,6 +476,7 @@ def test_completion_together_ai(): print("Cost for completion call together-computer/llama-2-70b: ", f"${float(cost):.10f}") except Exception as e: pytest.fail(f"Error occurred: {e}") + # test_completion_together_ai() # def test_customprompt_together_ai(): # try: @@ -650,7 +512,8 @@ def test_completion_bedrock_titan(): model="bedrock/amazon.titan-tg1-large", messages=messages, temperature=0.2, - max_tokens=20, + max_tokens=200, + top_p=0.8, logger_fn=logger_fn ) # Add any assertions here to check the response @@ -662,21 +525,20 @@ def test_completion_bedrock_titan(): def test_completion_bedrock_ai21(): try: + litellm.set_verbose = False response = completion( model="bedrock/ai21.j2-mid", messages=messages, temperature=0.2, - max_tokens=20, - logger_fn=logger_fn + top_p=0.2, + max_tokens=20 ) # Add any assertions here to check the response print(response) except Exception as e: pytest.fail(f"Error occurred: {e}") -# test_completion_bedrock_ai21() -# test_completion_sagemaker() ######## Test VLLM ######## # def test_completion_vllm(): # try: @@ -761,7 +623,7 @@ def test_completion_bedrock_ai21(): # # pass # except Exception as e: # pytest.fail(f"Error occurred: {e}") -# test_vertex_ai_stream() +# test_vertex_ai_stream() def test_completion_with_fallbacks(): diff --git a/litellm/tests/test_streaming.py b/litellm/tests/test_streaming.py index 6b1b2b9a143..10d72147805 100644 --- a/litellm/tests/test_streaming.py +++ b/litellm/tests/test_streaming.py @@ -24,6 +24,171 @@ def logger_fn(model_call_object: dict): user_message = "Hello, how are you?" messages = [{"content": user_message, "role": "user"}] + +first_openai_chunk_example = { + "id": "chatcmpl-7zSKLBVXnX9dwgRuDYVqVVDsgh2yp", + "object": "chat.completion.chunk", + "created": 1694881253, + "model": "gpt-4-0613", + "choices": [ + { + "index": 0, + "delta": { + "role": "assistant", + "content": "" + }, + "finish_reason": None # it's null + } + ] +} + +def validate_first_format(chunk): + # write a test to make sure chunk follows the same format as first_openai_chunk_example + assert isinstance(chunk, dict), "Chunk should be a dictionary." + assert "id" in chunk, "Chunk should have an 'id'." + assert isinstance(chunk['id'], str), "'id' should be a string." + + assert "object" in chunk, "Chunk should have an 'object'." + assert isinstance(chunk['object'], str), "'object' should be a string." + + assert "created" in chunk, "Chunk should have a 'created'." + assert isinstance(chunk['created'], int), "'created' should be an integer." + + assert "model" in chunk, "Chunk should have a 'model'." + assert isinstance(chunk['model'], str), "'model' should be a string." + + assert "choices" in chunk, "Chunk should have 'choices'." + assert isinstance(chunk['choices'], list), "'choices' should be a list." + + for choice in chunk['choices']: + assert isinstance(choice, dict), "Each choice should be a dictionary." + + assert "index" in choice, "Each choice should have 'index'." + assert isinstance(choice['index'], int), "'index' should be an integer." + + assert "delta" in choice, "Each choice should have 'delta'." + assert isinstance(choice['delta'], dict), "'delta' should be a dictionary." + + assert "role" in choice['delta'], "'delta' should have a 'role'." + assert isinstance(choice['delta']['role'], str), "'role' should be a string." + + assert "content" in choice['delta'], "'delta' should have 'content'." + assert isinstance(choice['delta']['content'], str), "'content' should be a string." + + assert "finish_reason" in choice, "Each choice should have 'finish_reason'." + assert (choice['finish_reason'] is None) or isinstance(choice['finish_reason'], str), "'finish_reason' should be None or a string." + +second_openai_chunk_example = { + "id": "chatcmpl-7zSKLBVXnX9dwgRuDYVqVVDsgh2yp", + "object": "chat.completion.chunk", + "created": 1694881253, + "model": "gpt-4-0613", + "choices": [ + { + "index": 0, + "delta": { + "content": "Hello" + }, + "finish_reason": None # it's null + } + ] +} + +def validate_second_format(chunk): + assert isinstance(chunk, dict), "Chunk should be a dictionary." + assert "id" in chunk, "Chunk should have an 'id'." + assert isinstance(chunk['id'], str), "'id' should be a string." + + assert "object" in chunk, "Chunk should have an 'object'." + assert isinstance(chunk['object'], str), "'object' should be a string." + + assert "created" in chunk, "Chunk should have a 'created'." + assert isinstance(chunk['created'], int), "'created' should be an integer." + + assert "model" in chunk, "Chunk should have a 'model'." + assert isinstance(chunk['model'], str), "'model' should be a string." + + assert "choices" in chunk, "Chunk should have 'choices'." + assert isinstance(chunk['choices'], list), "'choices' should be a list." + + for choice in chunk['choices']: + assert isinstance(choice, dict), "Each choice should be a dictionary." + + assert "index" in choice, "Each choice should have 'index'." + assert isinstance(choice['index'], int), "'index' should be an integer." + + assert "delta" in choice, "Each choice should have 'delta'." + assert isinstance(choice['delta'], dict), "'delta' should be a dictionary." + + assert "content" in choice['delta'], "'delta' should have 'content'." + assert isinstance(choice['delta']['content'], str), "'content' should be a string." + + assert "finish_reason" in choice, "Each choice should have 'finish_reason'." + assert (choice['finish_reason'] is None) or isinstance(choice['finish_reason'], str), "'finish_reason' should be None or a string." + +last_openai_chunk_example = { + "id": "chatcmpl-7zSKLBVXnX9dwgRuDYVqVVDsgh2yp", + "object": "chat.completion.chunk", + "created": 1694881253, + "model": "gpt-4-0613", + "choices": [ + { + "index": 0, + "delta": {}, + "finish_reason": "stop" + } + ] +} + +def validate_last_format(chunk): + assert isinstance(chunk, dict), "Chunk should be a dictionary." + assert "id" in chunk, "Chunk should have an 'id'." + assert isinstance(chunk['id'], str), "'id' should be a string." + + assert "object" in chunk, "Chunk should have an 'object'." + assert isinstance(chunk['object'], str), "'object' should be a string." + + assert "created" in chunk, "Chunk should have a 'created'." + assert isinstance(chunk['created'], int), "'created' should be an integer." + + assert "model" in chunk, "Chunk should have a 'model'." + assert isinstance(chunk['model'], str), "'model' should be a string." + + assert "choices" in chunk, "Chunk should have 'choices'." + assert isinstance(chunk['choices'], list), "'choices' should be a list." + + for choice in chunk['choices']: + assert isinstance(choice, dict), "Each choice should be a dictionary." + + assert "index" in choice, "Each choice should have 'index'." + assert isinstance(choice['index'], int), "'index' should be an integer." + + assert "delta" in choice, "Each choice should have 'delta'." + assert isinstance(choice['delta'], dict), "'delta' should be a dictionary." + + assert "finish_reason" in choice, "Each choice should have 'finish_reason'." + assert isinstance(choice['finish_reason'], str), "'finish_reason' should be a string." + +def streaming_format_tests(idx, chunk): + extracted_chunk = "" + finished = False + print(f"chunk: {chunk}") + if idx == 0: # ensure role assistant is set + validate_first_format(chunk=chunk) + role = chunk["choices"][0]["delta"]["role"] + assert role == "assistant" + elif idx == 1: # second chunk + validate_second_format(chunk=chunk) + if idx != 0: # ensure no role + if "role" in chunk["choices"][0]["delta"]: + raise Exception("role should not exist after first chunk") + if chunk["choices"][0]["finish_reason"]: # ensure finish reason is only in last chunk + validate_last_format(chunk=chunk) + finished = True + if "content" in chunk["choices"][0]["delta"]: + extracted_chunk = chunk["choices"][0]["delta"]["content"] + return extracted_chunk, finished + def test_completion_cohere_stream(): try: messages = [ @@ -38,36 +203,68 @@ def test_completion_cohere_stream(): ) complete_response = "" # Add any assertions here to check the response - for chunk in response: - print(f"chunk: {chunk}") - complete_response += chunk["choices"][0]["delta"]["content"] - if complete_response == "": + for idx, chunk in enumerate(response): + chunk, finished = streaming_format_tests(idx, chunk) + if finished: + break + complete_response += chunk + if complete_response.strip() == "": raise Exception("Empty response received") print(f"completion_response: {complete_response}") - except KeyError as e: - pass except Exception as e: pytest.fail(f"Error occurred: {e}") - -# test on baseten completion call -# try: -# response = completion( -# model="baseten/RqgAEn0", messages=messages, logger_fn=logger_fn -# ) -# print(f"response: {response}") -# complete_response = "" -# start_time = time.time() -# for chunk in response: -# chunk_time = time.time() -# print(f"time since initial request: {chunk_time - start_time:.5f}") -# print(chunk["choices"][0]["delta"]) -# complete_response += chunk["choices"][0]["delta"]["content"] -# if complete_response == "": -# raise Exception("Empty response received") -# print(f"complete response: {complete_response}") -# except: -# print(f"error occurred: {traceback.format_exc()}") -# pass + +def test_completion_claude_stream(): + try: + messages = [ + {"role": "system", "content": "You are a helpful assistant."}, + { + "role": "user", + "content": "how does a court case get to the Supreme Court?", + }, + ] + response = completion( + model="claude-instant-1", messages=messages, stream=True, max_tokens=50 + ) + complete_response = "" + # Add any assertions here to check the response + for idx, chunk in enumerate(response): + chunk, finished = streaming_format_tests(idx, chunk) + if finished: + break + complete_response += chunk + if complete_response.strip() == "": + raise Exception("Empty response received") + print(f"completion_response: {complete_response}") + except Exception as e: + pytest.fail(f"Error occurred: {e}") +# test_completion_claude_stream() + +def test_completion_bedrock_ai21_stream(): + try: + litellm.set_verbose = False + response = completion( + model="bedrock/amazon.titan-tg1-large", + messages=[{"role": "user", "content": "Be as verbose as possible and give as many details as possible, how does a court case get to the Supreme Court?"}], + temperature=1, + max_tokens=4096, + stream=True, + ) + complete_response = "" + # Add any assertions here to check the response + print(response) + for idx, chunk in enumerate(response): + chunk, finished = streaming_format_tests(idx, chunk) + if finished: + break + complete_response += chunk + if complete_response.strip() == "": + raise Exception("Empty response received") + except Exception as e: + pytest.fail(f"Error occurred: {e}") + + +# test_completion_cohere_stream() # test on openai completion call def test_openai_text_completion_call(): @@ -77,16 +274,15 @@ def test_openai_text_completion_call(): ) complete_response = "" start_time = time.time() - for chunk in response: - chunk_time = time.time() - print(f"chunk: {chunk}") - if "content" in chunk["choices"][0]["delta"]: - complete_response += chunk["choices"][0]["delta"]["content"] - if complete_response == "": + for idx, chunk in enumerate(response): + chunk, finished = streaming_format_tests(idx, chunk) + if finished: + break + complete_response += chunk + if complete_response.strip() == "": raise Exception("Empty response received") except: - print(f"error occurred: {traceback.format_exc()}") - pass + pytest.fail(f"error occurred: {traceback.format_exc()}") # # test on ai21 completion call def ai21_completion_call(): @@ -97,18 +293,18 @@ def ai21_completion_call(): print(f"response: {response}") complete_response = "" start_time = time.time() - for chunk in response: - chunk_time = time.time() - print(f"time since initial request: {chunk_time - start_time:.5f}") - print(chunk) - if "content" in chunk["choices"][0]["delta"]: - complete_response += chunk["choices"][0]["delta"]["content"] - if complete_response == "": + for idx, chunk in enumerate(response): + chunk, finished = streaming_format_tests(idx, chunk) + if finished: + break + complete_response += chunk + if complete_response.strip() == "": raise Exception("Empty response received") + print(f"completion_response: {complete_response}") except: - print(f"error occurred: {traceback.format_exc()}") - pass + pytest.fail(f"error occurred: {traceback.format_exc()}") +# ai21_completion_call() # test on openai completion call def test_openai_chat_completion_call(): try: @@ -117,107 +313,20 @@ def test_openai_chat_completion_call(): ) complete_response = "" start_time = time.time() - for chunk in response: - print(chunk) - if chunk["choices"][0]["finish_reason"]: + for idx, chunk in enumerate(response): + chunk, finished = streaming_format_tests(idx, chunk) + if finished: break - # if chunk["choices"][0]["delta"]["role"] != "assistant": - # raise Exception("invalid role") - if "content" in chunk["choices"][0]["delta"]: - complete_response += chunk["choices"][0]["delta"]["content"] + complete_response += chunk # print(f'complete_chunk: {complete_response}') if complete_response.strip() == "": raise Exception("Empty response received") + print(f"complete response: {complete_response}") except: print(f"error occurred: {traceback.format_exc()}") pass -test_openai_chat_completion_call() -async def completion_call(): - try: - response = completion( - model="gpt-3.5-turbo", messages=messages, stream=True, logger_fn=logger_fn - ) - print(f"response: {response}") - complete_response = "" - start_time = time.time() - # Change for loop to async for loop - async for chunk in response: - chunk_time = time.time() - print(f"time since initial request: {chunk_time - start_time:.5f}") - print(chunk["choices"][0]["delta"]) - if "content" in chunk["choices"][0]["delta"]: - complete_response += chunk["choices"][0]["delta"]["content"] - if complete_response == "": - raise Exception("Empty response received") - except: - print(f"error occurred: {traceback.format_exc()}") - pass - -# asyncio.run(completion_call()) - -# # test on azure completion call -# try: -# response = completion( -# model="azure/chatgpt-test", messages=messages, stream=True, logger_fn=logger_fn -# ) -# response = "" -# start_time = time.time() -# for chunk in response: -# chunk_time = time.time() -# print(f"time since initial request: {chunk_time - start_time:.2f}") -# print(chunk["choices"][0]["delta"]) -# response += chunk["choices"][0]["delta"] -# if response == "": -# raise Exception("Empty response received") -# except: -# print(f"error occurred: {traceback.format_exc()}") -# pass - - -# # test on huggingface completion call -# try: -# start_time = time.time() -# response = completion( -# model="gpt-3.5-turbo", messages=messages, stream=True, logger_fn=logger_fn -# ) -# complete_response = "" -# for chunk in response: -# chunk_time = time.time() -# print(f"time since initial request: {chunk_time - start_time:.2f}") -# print(chunk["choices"][0]["delta"]) -# complete_response += chunk["choices"][0]["delta"]["content"] if len(chunk["choices"][0]["delta"].keys()) > 0 else "" -# if complete_response == "": -# raise Exception("Empty response received") -# except: -# print(f"error occurred: {traceback.format_exc()}") -# pass - -# test on together ai completion call - replit-code-3b -def test_together_ai_completion_call_replit(): - try: - start_time = time.time() - response = completion( - model="Replit-Code-3B", messages=messages, logger_fn=logger_fn, stream=True - ) - complete_response = "" - print(f"returned response object: {response}") - for chunk in response: - chunk_time = time.time() - print(f"time since initial request: {chunk_time - start_time:.2f}") - print(chunk["choices"][0]["delta"]) - complete_response += ( - chunk["choices"][0]["delta"]["content"] - if len(chunk["choices"][0]["delta"].keys()) > 0 - else "" - ) - if complete_response == "": - raise Exception("Empty response received") - except KeyError as e: - pass - except: - print(f"error occurred: {traceback.format_exc()}") - pass +# test_openai_chat_completion_call() # # test on together ai completion call - starcoder def test_together_ai_completion_call_starcoder(): @@ -231,50 +340,54 @@ def test_together_ai_completion_call_starcoder(): ) complete_response = "" print(f"returned response object: {response}") - for chunk in response: - chunk_time = time.time() - complete_response += ( - chunk["choices"][0]["delta"]["content"] - if len(chunk["choices"][0]["delta"].keys()) > 0 - else "" - ) - if len(complete_response) > 0: - print(complete_response) + for idx, chunk in enumerate(response): + chunk, finished = streaming_format_tests(idx, chunk) + if finished: + break + complete_response += chunk if complete_response == "": raise Exception("Empty response received") - except KeyError as e: - pass + print(f"complete response: {complete_response}") except: print(f"error occurred: {traceback.format_exc()}") pass -# test on aleph alpha completion call - commented out as it's expensive to run this on circle ci for every build -# def test_aleph_alpha_call(): -# try: -# start_time = time.time() -# response = completion( -# model="luminous-base", -# messages=messages, -# logger_fn=logger_fn, -# stream=True, -# ) -# complete_response = "" -# print(f"returned response object: {response}") -# for chunk in response: -# chunk_time = time.time() -# complete_response += ( -# chunk["choices"][0]["delta"]["content"] -# if len(chunk["choices"][0]["delta"].keys()) > 0 -# else "" -# ) -# if len(complete_response) > 0: -# print(complete_response) -# if complete_response == "": -# raise Exception("Empty response received") -# except: -# print(f"error occurred: {traceback.format_exc()}") -# pass -#### Test Async streaming +#### Test Function calling + streaming #### + +def test_completion_openai_with_functions(): + function1 = [ + { + "name": "get_current_weather", + "description": "Get the current weather in a given location", + "parameters": { + "type": "object", + "properties": { + "location": { + "type": "string", + "description": "The city and state, e.g. San Francisco, CA", + }, + "unit": {"type": "string", "enum": ["celsius", "fahrenheit"]}, + }, + "required": ["location"], + }, + } + ] + try: + response = completion( + model="gpt-3.5-turbo", messages=messages, functions=function1, stream=True + ) + # Add any assertions here to check the response + print(response) + for chunk in response: + print(chunk) + if chunk["choices"][0]["finish_reason"] == "stop": + break + print(chunk["choices"][0]["finish_reason"]) + print(chunk["choices"][0]["delta"]["content"]) + except Exception as e: + pytest.fail(f"Error occurred: {e}") + +#### Test Async streaming #### # # test on ai21 completion call async def ai21_async_completion_call(): @@ -286,13 +399,305 @@ async def ai21_async_completion_call(): complete_response = "" start_time = time.time() # Change for loop to async for loop + idx = 0 async for chunk in response: - chunk_time = time.time() - print(f"time since initial request: {chunk_time - start_time:.5f}") - print(chunk["choices"][0]["delta"]) - complete_response += chunk["choices"][0]["delta"]["content"] - if complete_response == "": + chunk, finished = streaming_format_tests(idx, chunk) + if finished: + break + complete_response += chunk + idx += 1 + if complete_response.strip() == "": raise Exception("Empty response received") + print(f"complete response: {complete_response}") except: print(f"error occurred: {traceback.format_exc()}") - pass \ No newline at end of file + pass + +# asyncio.run(ai21_async_completion_call()) + +async def completion_call(): + try: + response = completion( + model="gpt-3.5-turbo", messages=messages, stream=True, logger_fn=logger_fn + ) + print(f"response: {response}") + complete_response = "" + start_time = time.time() + # Change for loop to async for loop + idx = 0 + async for chunk in response: + chunk, finished = streaming_format_tests(idx, chunk) + if finished: + break + complete_response += chunk + idx += 1 + if complete_response.strip() == "": + raise Exception("Empty response received") + print(f"complete response: {complete_response}") + except: + print(f"error occurred: {traceback.format_exc()}") + pass + +# asyncio.run(completion_call()) + +#### Test Function Calling + Streaming #### + +final_openai_function_call_example = { + "id": "chatcmpl-7zVNA4sXUftpIg6W8WlntCyeBj2JY", + "object": "chat.completion", + "created": 1694892960, + "model": "gpt-3.5-turbo-0613", + "choices": [ + { + "index": 0, + "message": { + "role": "assistant", + "content": None, + "function_call": { + "name": "get_current_weather", + "arguments": "{\n \"location\": \"Boston, MA\"\n}" + } + }, + "finish_reason": "function_call" + } + ], + "usage": { + "prompt_tokens": 82, + "completion_tokens": 18, + "total_tokens": 100 + } +} + +function_calling_output_structure = { + "id": str, + "object": str, + "created": int, + "model": str, + "choices": [ + { + "index": int, + "message": { + "role": str, + "content": (type(None), str), + "function_call": { + "name": str, + "arguments": str + } + }, + "finish_reason": str + } + ], + "usage": { + "prompt_tokens": int, + "completion_tokens": int, + "total_tokens": int + } + } + +def validate_final_structure(item, structure=function_calling_output_structure): + if isinstance(item, list): + if not all(validate_final_structure(i, structure[0]) for i in item): + return Exception("Function calling final output doesn't match expected output format") + elif isinstance(item, dict): + if not all(k in item and validate_final_structure(item[k], v) for k, v in structure.items()): + return Exception("Function calling final output doesn't match expected output format") + else: + if not isinstance(item, structure): + return Exception("Function calling final output doesn't match expected output format") + return True + + +first_openai_function_call_example = { + "id": "chatcmpl-7zVRoE5HjHYsCMaVSNgOjzdhbS3P0", + "object": "chat.completion.chunk", + "created": 1694893248, + "model": "gpt-3.5-turbo-0613", + "choices": [ + { + "index": 0, + "delta": { + "role": "assistant", + "content": None, + "function_call": { + "name": "get_current_weather", + "arguments": "" + } + }, + "finish_reason": None + } + ] +} + +def validate_first_function_call_chunk_structure(item): + if not isinstance(item, dict): + raise Exception("Incorrect format") + + required_keys = {"id", "object", "created", "model", "choices"} + for key in required_keys: + if key not in item: + raise Exception("Incorrect format") + + if not isinstance(item["choices"], list) or not item["choices"]: + raise Exception("Incorrect format") + + required_keys_in_choices_array = {"index", "delta", "finish_reason"} + for choice in item["choices"]: + if not isinstance(choice, dict): + raise Exception("Incorrect format") + for key in required_keys_in_choices_array: + if key not in choice: + raise Exception("Incorrect format") + + if not isinstance(choice["delta"], dict): + raise Exception("Incorrect format") + + required_keys_in_delta = {"role", "content", "function_call"} + for key in required_keys_in_delta: + if key not in choice["delta"]: + raise Exception("Incorrect format") + + if not isinstance(choice["delta"]["function_call"], dict): + raise Exception("Incorrect format") + + required_keys_in_function_call = {"name", "arguments"} + for key in required_keys_in_function_call: + if key not in choice["delta"]["function_call"]: + raise Exception("Incorrect format") + + return True + +second_function_call_chunk_format = { + "id": "chatcmpl-7zVRoE5HjHYsCMaVSNgOjzdhbS3P0", + "object": "chat.completion.chunk", + "created": 1694893248, + "model": "gpt-3.5-turbo-0613", + "choices": [ + { + "index": 0, + "delta": { + "function_call": { + "arguments": "{\n" + } + }, + "finish_reason": None + } + ] +} + + +def validate_second_function_call_chunk_structure(data): + if not isinstance(data, dict): + raise Exception("Incorrect format") + + required_keys = {"id", "object", "created", "model", "choices"} + for key in required_keys: + if key not in data: + raise Exception("Incorrect format") + + if not isinstance(data["choices"], list) or not data["choices"]: + raise Exception("Incorrect format") + + required_keys_in_choices_array = {"index", "delta", "finish_reason"} + for choice in data["choices"]: + if not isinstance(choice, dict): + raise Exception("Incorrect format") + for key in required_keys_in_choices_array: + if key not in choice: + raise Exception("Incorrect format") + + if "function_call" not in choice["delta"] or "arguments" not in choice["delta"]["function_call"]: + raise Exception("Incorrect format") + + return True + + +final_function_call_chunk_example = { + "id": "chatcmpl-7zVRoE5HjHYsCMaVSNgOjzdhbS3P0", + "object": "chat.completion.chunk", + "created": 1694893248, + "model": "gpt-3.5-turbo-0613", + "choices": [ + { + "index": 0, + "delta": {}, + "finish_reason": "function_call" + } + ] +} + + +def validate_final_function_call_chunk_structure(data): + if not isinstance(data, dict): + raise Exception("Incorrect format") + + required_keys = {"id", "object", "created", "model", "choices"} + for key in required_keys: + if key not in data: + raise Exception("Incorrect format") + + if not isinstance(data["choices"], list) or not data["choices"]: + raise Exception("Incorrect format") + + required_keys_in_choices_array = {"index", "delta", "finish_reason"} + for choice in data["choices"]: + if not isinstance(choice, dict): + raise Exception("Incorrect format") + for key in required_keys_in_choices_array: + if key not in choice: + raise Exception("Incorrect format") + + return True + +def streaming_and_function_calling_format_tests(idx, chunk): + extracted_chunk = "" + finished = False + print(f"idx: {idx}") + print(f"chunk: {chunk}") + decision = False + if idx == 0: # ensure role assistant is set + decision = validate_first_function_call_chunk_structure(chunk) + role = chunk["choices"][0]["delta"]["role"] + assert role == "assistant" + elif idx != 0: # second chunk + try: + decision = validate_second_function_call_chunk_structure(data=chunk) + except: # check if it's the last chunk (returns an empty delta {} ) + decision = validate_final_function_call_chunk_structure(data=chunk) + finished = True + if "content" in chunk["choices"][0]["delta"]: + extracted_chunk = chunk["choices"][0]["delta"]["content"] + if decision == False: + raise Exception("incorrect format") + return extracted_chunk, finished + +def test_openai_streaming_and_function_calling(): + function1 = [ + { + "name": "get_current_weather", + "description": "Get the current weather in a given location", + "parameters": { + "type": "object", + "properties": { + "location": { + "type": "string", + "description": "The city and state, e.g. San Francisco, CA", + }, + "unit": {"type": "string", "enum": ["celsius", "fahrenheit"]}, + }, + "required": ["location"], + }, + } + ] + messages=[{"role": "user", "content": "What is the weather like in Boston?"}] + try: + response = completion( + model="gpt-3.5-turbo", functions=function1, messages=messages, stream=True + ) + # Add any assertions here to check the response + for idx, chunk in enumerate(response): + streaming_and_function_calling_format_tests(idx=idx, chunk=chunk) + except Exception as e: + pytest.fail(f"Error occurred: {e}") + raise e + +# test_openai_streaming_and_function_calling() diff --git a/litellm/utils.py b/litellm/utils.py index 6ee25ac486b..9431faa3b87 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -80,6 +80,8 @@ last_fetched_at_keys = None # 'usage': {'prompt_tokens': 18, 'completion_tokens': 23, 'total_tokens': 41} # } +def _generate_id(): # private helper function + return 'chatcmpl-' + str(uuid.uuid4()) class Message(OpenAIObject): def __init__(self, content="default", role="assistant", logprobs=None, **params): @@ -89,9 +91,9 @@ class Message(OpenAIObject): self.logprobs = logprobs class Delta(OpenAIObject): - def __init__(self, content="", logprobs=None, role=None, **params): + def __init__(self, content=None, logprobs=None, role=None, **params): super(Delta, self).__init__(**params) - if content != "": + if content is not None: self.content = content if role: self.role = role @@ -105,20 +107,34 @@ class Choices(OpenAIObject): self.message = message class StreamingChoices(OpenAIObject): - def __init__(self, finish_reason=None, index=0, delta=Delta(), **params): + def __init__(self, finish_reason=None, index=0, delta: Optional[Delta]=None, **params): super(StreamingChoices, self).__init__(**params) self.finish_reason = finish_reason self.index = index - self.delta = delta + if delta: + self.delta = delta + else: + self.delta = Delta() class ModelResponse(OpenAIObject): - def __init__(self, choices=None, created=None, model=None, usage=None, stream=False, **params): - super(ModelResponse, self).__init__(**params) + def __init__(self, id=None, choices=None, created=None, model=None, usage=None, stream=False, **params): if stream: - self.choices = self.choices = choices if choices else [StreamingChoices()] + self.object = "chat.completion.chunk" + self.choices = [StreamingChoices()] else: + if model in litellm.open_ai_embedding_models: + self.object = "embedding" + else: + self.object = "chat.completion" self.choices = self.choices = choices if choices else [Choices()] - self.created = created + if id is None: + self.id = _generate_id() + else: + self.id = id + if created is None: + self.created = int(time.time()) + else: + self.created = created self.model = model self.usage = ( usage @@ -129,6 +145,7 @@ class ModelResponse(OpenAIObject): "total_tokens": None, } ) + super(ModelResponse, self).__init__(**params) def to_dict_recursive(self): d = super().to_dict_recursive() @@ -811,6 +828,7 @@ def get_optional_params( # use the openai defaults model=None, custom_llm_provider="", top_k=40, + return_full_text=False, task=None ): optional_params = {} @@ -868,9 +886,10 @@ def get_optional_params( # use the openai defaults optional_params["max_new_tokens"] = max_tokens if presence_penalty != 0: optional_params["repetition_penalty"] = presence_penalty + optional_params["return_full_text"] = return_full_text optional_params["details"] = True optional_params["task"] = task - elif custom_llm_provider == "together_ai" or ("togethercomputer" in model): + elif custom_llm_provider == "together_ai": if stream: optional_params["stream_tokens"] = stream if temperature != 1: @@ -931,6 +950,30 @@ def get_optional_params( # use the openai defaults optional_params["temperature"] = temperature if top_p != 1: optional_params["top_p"] = top_p + elif custom_llm_provider == "bedrock": + if "ai21" in model or "anthropic" in model: + # params "maxTokens":200,"temperature":0,"topP":250,"stop_sequences":[], + # https://us-west-2.console.aws.amazon.com/bedrock/home?region=us-west-2#/providers?model=j2-ultra + if max_tokens != float("inf"): + optional_params["maxTokens"] = max_tokens + if temperature != 1: + optional_params["temperature"] = temperature + if stop != None: + optional_params["stop_sequences"] = stop + if top_p != 1: + optional_params["topP"] = top_p + + elif "amazon" in model: # amazon titan llms + # see https://us-west-2.console.aws.amazon.com/bedrock/home?region=us-west-2#/providers?model=titan-large + if max_tokens != float("inf"): + optional_params["maxTokenCount"] = max_tokens + if temperature != 1: + optional_params["temperature"] = temperature + if stop != None: + optional_params["stopSequences"] = stop + if top_p != 1: + optional_params["topP"] = top_p + elif model in litellm.aleph_alpha_models: if max_tokens != float("inf"): optional_params["maximum_tokens"] = max_tokens @@ -1017,8 +1060,10 @@ def get_llm_provider(model: str, custom_llm_provider: Optional[str] = None): # check if model in known model provider list ## openai - chatcompletion + text completion - if model in litellm.open_ai_chat_completion_models or model in litellm.open_ai_text_completion_models: + if model in litellm.open_ai_chat_completion_models: custom_llm_provider = "openai" + elif model in litellm.open_ai_text_completion_models: + custom_llm_provider = "text-completion-openai" ## anthropic elif model in litellm.anthropic_models: custom_llm_provider = "anthropic" @@ -1041,7 +1086,7 @@ def get_llm_provider(model: str, custom_llm_provider: Optional[str] = None): elif model in litellm.ai21_models: custom_llm_provider = "ai21" ## together_ai - elif model in litellm.together_ai_models or "togethercomputer": + elif model in litellm.together_ai_models: custom_llm_provider = "together_ai" ## aleph_alpha elif model in litellm.aleph_alpha_models: @@ -2335,6 +2380,7 @@ class CustomStreamWrapper: self.custom_llm_provider = custom_llm_provider self.logging_obj = logging_obj self.completion_stream = completion_stream + self.sent_first_chunk = False if self.logging_obj: # Log the type of the received item self.logging_obj.post_call(str(type(completion_stream))) @@ -2406,13 +2452,13 @@ class CustomStreamWrapper: chunk = chunk.decode("utf-8") data_json = json.loads(chunk) try: - print(f"data json: {data_json}") return data_json["text"] except: raise ValueError(f"Unable to parse response. Original response: {chunk}") def handle_openai_text_completion_chunk(self, chunk): try: + print(f"chunk: {chunk}") return chunk["choices"][0]["text"] except: raise ValueError(f"Unable to parse response. Original response: {chunk}") @@ -2451,74 +2497,84 @@ class CustomStreamWrapper: traceback.print_exc() return "" + def handle_bedrock_stream(self): + if self.completion_stream: + event = next(self.completion_stream) + chunk = event.get('chunk') + if chunk: + chunk_data = json.loads(chunk.get('bytes').decode()) + return chunk_data['outputText'] + return "" + + ## needs to handle the empty string case (even starting chunk can be an empty string) def __next__(self): + model_response = ModelResponse(stream=True, model=self.model) try: - # return this for all models - completion_obj = {"content": ""} # default to role being assistant - if self.model in litellm.anthropic_models: - chunk = next(self.completion_stream) - completion_obj["content"] = self.handle_anthropic_chunk(chunk) - elif self.model == "replicate" or self.custom_llm_provider == "replicate": - chunk = next(self.completion_stream) - completion_obj["content"] = chunk - elif ( - self.custom_llm_provider and self.custom_llm_provider == "together_ai" - ) or ("togethercomputer" in self.model): - chunk = next(self.completion_stream) - text_data = self.handle_together_ai_chunk(chunk) - if text_data == "": - return self.__next__() - completion_obj["content"] = text_data - elif self.custom_llm_provider and self.custom_llm_provider == "huggingface": - chunk = next(self.completion_stream) - completion_obj["content"] = self.handle_huggingface_chunk(chunk) - elif self.custom_llm_provider and self.custom_llm_provider == "baseten": # baseten doesn't provide streaming - chunk = next(self.completion_stream) - completion_obj["content"] = self.handle_baseten_chunk(chunk) - elif self.custom_llm_provider and self.custom_llm_provider == "ai21": #ai21 doesn't provide streaming - chunk = next(self.completion_stream) - completion_obj["content"] = self.handle_ai21_chunk(chunk) - elif self.custom_llm_provider and self.custom_llm_provider == "vllm": - chunk = next(self.completion_stream) - completion_obj["content"] = chunk[0].outputs[0].text - elif self.model in litellm.aleph_alpha_models: #aleph alpha doesn't provide streaming - chunk = next(self.completion_stream) - completion_obj["content"] = self.handle_aleph_alpha_chunk(chunk) - elif self.model in litellm.open_ai_text_completion_models: - chunk = next(self.completion_stream) - completion_obj["content"] = self.handle_openai_text_completion_chunk(chunk) - elif self.model in litellm.nlp_cloud_models or self.custom_llm_provider == "nlp_cloud": - chunk = next(self.completion_stream) - completion_obj["content"] = self.handle_nlp_cloud_chunk(chunk) - elif self.model in (litellm.vertex_chat_models + litellm.vertex_code_chat_models + litellm.vertex_text_models + litellm.vertex_code_text_models): - chunk = next(self.completion_stream) - completion_obj["content"] = str(chunk) - elif self.model in litellm.cohere_models or self.custom_llm_provider == "cohere": - chunk = next(self.completion_stream) - completion_obj["content"] = self.handle_cohere_chunk(chunk) - else: # openai chat/azure models - chunk = next(self.completion_stream) - model_response = chunk + while True: # loop until a non-empty string is found + # return this for all models + completion_obj = {"content": ""} + if self.custom_llm_provider and self.custom_llm_provider == "anthropic": + chunk = next(self.completion_stream) + completion_obj["content"] = self.handle_anthropic_chunk(chunk) + elif self.model == "replicate" or self.custom_llm_provider == "replicate": + chunk = next(self.completion_stream) + completion_obj["content"] = chunk + elif ( + self.custom_llm_provider and self.custom_llm_provider == "together_ai"): + chunk = next(self.completion_stream) + text_data = self.handle_together_ai_chunk(chunk) + if text_data == "": + return self.__next__() + completion_obj["content"] = text_data + elif self.custom_llm_provider and self.custom_llm_provider == "huggingface": + chunk = next(self.completion_stream) + completion_obj["content"] = self.handle_huggingface_chunk(chunk) + elif self.custom_llm_provider and self.custom_llm_provider == "baseten": # baseten doesn't provide streaming + chunk = next(self.completion_stream) + completion_obj["content"] = self.handle_baseten_chunk(chunk) + elif self.custom_llm_provider and self.custom_llm_provider == "ai21": #ai21 doesn't provide streaming + chunk = next(self.completion_stream) + completion_obj["content"] = self.handle_ai21_chunk(chunk) + elif self.custom_llm_provider and self.custom_llm_provider == "vllm": + chunk = next(self.completion_stream) + completion_obj["content"] = chunk[0].outputs[0].text + elif self.custom_llm_provider and self.custom_llm_provider == "aleph-alpha": #aleph alpha doesn't provide streaming + chunk = next(self.completion_stream) + completion_obj["content"] = self.handle_aleph_alpha_chunk(chunk) + elif self.custom_llm_provider and self.custom_llm_provider == "text-completion-openai": + chunk = next(self.completion_stream) + completion_obj["content"] = self.handle_openai_text_completion_chunk(chunk) + elif self.model in litellm.nlp_cloud_models or self.custom_llm_provider == "nlp_cloud": + chunk = next(self.completion_stream) + completion_obj["content"] = self.handle_nlp_cloud_chunk(chunk) + elif self.model in (litellm.vertex_chat_models + litellm.vertex_code_chat_models + litellm.vertex_text_models + litellm.vertex_code_text_models): + chunk = next(self.completion_stream) + completion_obj["content"] = str(chunk) + elif self.custom_llm_provider == "cohere": + chunk = next(self.completion_stream) + completion_obj["content"] = self.handle_cohere_chunk(chunk) + elif self.custom_llm_provider == "bedrock": + completion_obj["content"] = self.handle_bedrock_stream() + else: # openai chat/azure models + chunk = next(self.completion_stream) + model_response = chunk + # LOGGING + threading.Thread(target=self.logging_obj.success_handler, args=(completion_obj,)).start() + return model_response + # LOGGING threading.Thread(target=self.logging_obj.success_handler, args=(completion_obj,)).start() - return model_response - - # LOGGING - threading.Thread(target=self.logging_obj.success_handler, args=(completion_obj,)).start() - model_response = ModelResponse(stream=True) - model_response.choices[0].delta = completion_obj - model_response.model = self.model - - if model_response.choices[0].delta['content'] == "": - model_response.choices[0].delta = { - "content": completion_obj["content"], - } - return model_response + model_response.model = self.model + if len(completion_obj["content"]) > 0: # cannot set content of an OpenAI Object to be an empty string + if self.sent_first_chunk == False: + completion_obj["role"] = "assistant" + self.sent_first_chunk = True + model_response.choices[0].delta = Delta(**completion_obj) + return model_response except StopIteration: raise StopIteration except Exception as e: - print(e) - model_response = ModelResponse(stream=True) + traceback.print_exc() model_response.choices[0].finish_reason = "stop" return model_response diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index c19df784e72..69fde5211c9 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -1,4 +1,19 @@ { + "gpt-4": { + "max_tokens": 8192, + "input_cost_per_token": 0.000003, + "output_cost_per_token": 0.00006 + }, + "gpt-4-0613": { + "max_tokens": 8192, + "input_cost_per_token": 0.000003, + "output_cost_per_token": 0.00006 + }, + "gpt-4-32k": { + "max_tokens": 32768, + "input_cost_per_token": 0.00006, + "output_cost_per_token": 0.00012 + }, "gpt-3.5-turbo": { "max_tokens": 4097, "input_cost_per_token": 0.0000015, @@ -24,21 +39,6 @@ "input_cost_per_token": 0.000003, "output_cost_per_token": 0.000004 }, - "gpt-4": { - "max_tokens": 8192, - "input_cost_per_token": 0.000003, - "output_cost_per_token": 0.00006 - }, - "gpt-4-0613": { - "max_tokens": 8192, - "input_cost_per_token": 0.000003, - "output_cost_per_token": 0.00006 - }, - "gpt-4-32k": { - "max_tokens": 32768, - "input_cost_per_token": 0.00006, - "output_cost_per_token": 0.00012 - }, "claude-instant-1": { "max_tokens": 100000, "input_cost_per_token": 0.00000163, diff --git a/proxy-server/.DS_Store b/proxy-server/.DS_Store index 739982f1422..7a42e831cf5 100644 Binary files a/proxy-server/.DS_Store and b/proxy-server/.DS_Store differ diff --git a/pyproject.toml b/pyproject.toml index 6c8acc6f8a2..a41b1815874 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [tool.poetry] name = "litellm" -version = "0.1.674" +version = "0.1.687" description = "Library to easily interface with LLM API providers" authors = ["BerriAI"] license = "MIT License"