mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
Merge branch 'main' of https://github.com/WilliamEspegren/litellm
This commit is contained in:
commit
14fd6770be
24 changed files with 908 additions and 495 deletions
3
.gitmodules
vendored
Normal file
3
.gitmodules
vendored
Normal file
|
|
@ -0,0 +1,3 @@
|
|||
[submodule "fastrepl-proxy"]
|
||||
path = fastrepl-proxy
|
||||
url = https://github.com/BerriAI/litellm
|
||||
|
|
@ -1,4 +1,4 @@
|
|||
# Mock Requests - Save Testing Costs 💰
|
||||
# Mock Completion() Responses - Save Testing Costs 💰
|
||||
|
||||
For testing purposes, you can use `completion()` with `mock_response` to mock calling the completion endpoint.
|
||||
|
||||
|
|
|
|||
34
docs/my-website/docs/max_tokens_cost.md
Normal file
34
docs/my-website/docs/max_tokens_cost.md
Normal file
|
|
@ -0,0 +1,34 @@
|
|||
# get context window & cost per token
|
||||
|
||||
100+ LLMs supported: See full list [here](https://github.com/BerriAI/litellm/blob/main/model_prices_and_context_window.json)
|
||||
|
||||
## using api.litellm.ai
|
||||
```shell
|
||||
curl 'https://api.litellm.ai/get_max_tokens?model=claude-2'
|
||||
```
|
||||
|
||||
### output
|
||||
```json
|
||||
{
|
||||
"input_cost_per_token": 1.102e-05,
|
||||
"max_tokens": 100000,
|
||||
"model": "claude-2",
|
||||
"output_cost_per_token": 3.268e-05
|
||||
}
|
||||
```
|
||||
|
||||
## using the litellm python package
|
||||
```python
|
||||
import litellm
|
||||
model_data = litellm.model_cost["gpt-4"]
|
||||
```
|
||||
|
||||
### output
|
||||
```json
|
||||
{
|
||||
"input_cost_per_token": 3e-06,
|
||||
"max_tokens": 8192,
|
||||
"model": "gpt-4",
|
||||
"output_cost_per_token": 6e-05
|
||||
}
|
||||
```
|
||||
|
|
@ -1,13 +1,13 @@
|
|||
# AWS Bedrock
|
||||
|
||||
### API KEYS
|
||||
## API KEYS
|
||||
```python
|
||||
os.environ["AWS_ACCESS_KEY_ID"] = ""
|
||||
os.environ["AWS_SECRET_ACCESS_KEY"] = ""
|
||||
os.environ["AWS_REGION_NAME"] = ""
|
||||
```
|
||||
|
||||
### Usage
|
||||
## Usage
|
||||
```python
|
||||
import os
|
||||
from litellm import completion
|
||||
|
|
@ -24,7 +24,7 @@ response = completion(
|
|||
)
|
||||
```
|
||||
|
||||
### Supported AWS Bedrock Models
|
||||
## Supported AWS Bedrock Models
|
||||
Here's an example of using a bedrock model with LiteLLM
|
||||
|
||||
| Model Name | Function Call | Required OS Variables |
|
||||
|
|
@ -33,7 +33,54 @@ Here's an example of using a bedrock model with LiteLLM
|
|||
| AI21 J2-Ultra | `completion(model='bedrock/ai21.j2-ultra', messages=messages)` | `os.environ['AWS_ACCESS_KEY_ID']`, `os.environ['AWS_SECRET_ACCESS_KEY']`, `os.environ['AWS_REGION_NAME']` |
|
||||
| AI21 J2-Mid | `completion(model='bedrock/ai21.j2-mid', messages=messages)` | `os.environ['AWS_ACCESS_KEY_ID']`, `os.environ['AWS_SECRET_ACCESS_KEY']`, `os.environ['AWS_REGION_NAME']` |
|
||||
|
||||
### Troubleshooting
|
||||
## Streaming
|
||||
Bedrock currently supports streaming for the following llms:
|
||||
* `bedrock/amazon.titan-tg1-large`
|
||||
|
||||
Example Usage
|
||||
```python
|
||||
import os
|
||||
from litellm import completion
|
||||
|
||||
os.environ["AWS_ACCESS_KEY_ID"] = ""
|
||||
os.environ["AWS_SECRET_ACCESS_KEY"] = ""
|
||||
os.environ["AWS_REGION_NAME"] = ""
|
||||
|
||||
response = completion(
|
||||
model="bedrock/amazon.titan-tg1-large",
|
||||
messages=[{ "content": "Hello, how are you?","role": "user"}],
|
||||
temperature=0.2,
|
||||
max_tokens=80,
|
||||
stream=True
|
||||
)
|
||||
|
||||
for chunk in response:
|
||||
print(chunk)
|
||||
```
|
||||
|
||||
### Example Streaming Output Chunk
|
||||
```json
|
||||
{
|
||||
"choices": [
|
||||
{
|
||||
"finish_reason": null,
|
||||
"index": 0,
|
||||
"delta": {
|
||||
"content": "ase can appeal the case to a higher federal court. If a higher federal court rules in a way that conflicts with a ruling from a lower federal court or conflicts with a ruling from a higher state court, the parties involved in the case can appeal the case to the Supreme Court. In order to appeal a case to the Sup"
|
||||
}
|
||||
}
|
||||
],
|
||||
"created": null,
|
||||
"model": "amazon.titan-tg1-large",
|
||||
"usage": {
|
||||
"prompt_tokens": null,
|
||||
"completion_tokens": null,
|
||||
"total_tokens": null
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
## Troubleshooting
|
||||
If creating a boto3 bedrock client fails with `Unknown service: 'bedrock'`
|
||||
Try re installing boto3 using the following commands
|
||||
```shell
|
||||
|
|
|
|||
|
|
@ -79,6 +79,7 @@ const sidebars = {
|
|||
"token_usage",
|
||||
"exception_mapping",
|
||||
'debugging/local_debugging',
|
||||
"max_tokens_cost",
|
||||
"budget_manager",
|
||||
"proxy_api",
|
||||
{
|
||||
|
|
|
|||
1
fastrepl-proxy
Submodule
1
fastrepl-proxy
Submodule
|
|
@ -0,0 +1 @@
|
|||
Subproject commit f2fe83e002a7c3ddedf4e500665644adfd31b9fc
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
|
|
@ -59,6 +59,7 @@ def completion(
|
|||
encoding,
|
||||
logging_obj,
|
||||
optional_params=None,
|
||||
stream=False,
|
||||
litellm_params=None,
|
||||
logger_fn=None,
|
||||
):
|
||||
|
|
@ -94,14 +95,8 @@ def completion(
|
|||
else: # amazon titan
|
||||
data = json.dumps({
|
||||
"inputText": prompt,
|
||||
"textGenerationConfig":{
|
||||
"maxTokenCount":4096,
|
||||
"stopSequences":[],
|
||||
"temperature":0,
|
||||
"topP":0.9
|
||||
}
|
||||
"textGenerationConfig": optional_params,
|
||||
})
|
||||
|
||||
## LOGGING
|
||||
logging_obj.pre_call(
|
||||
input=prompt,
|
||||
|
|
@ -112,6 +107,15 @@ def completion(
|
|||
## COMPLETION CALL
|
||||
accept = 'application/json'
|
||||
contentType = 'application/json'
|
||||
if stream == True:
|
||||
response = client.invoke_model_with_response_stream(
|
||||
body=data,
|
||||
modelId=model,
|
||||
accept=accept,
|
||||
contentType=contentType
|
||||
)
|
||||
response = response.get('body')
|
||||
return response
|
||||
|
||||
response = client.invoke_model(
|
||||
body=data,
|
||||
|
|
@ -120,50 +124,48 @@ def completion(
|
|||
contentType=contentType
|
||||
)
|
||||
response_body = json.loads(response.get('body').read())
|
||||
if "stream" in optional_params and optional_params["stream"] == True:
|
||||
return response.iter_lines()
|
||||
else:
|
||||
## LOGGING
|
||||
logging_obj.post_call(
|
||||
input=prompt,
|
||||
api_key="",
|
||||
original_response=response,
|
||||
additional_args={"complete_input_dict": data},
|
||||
)
|
||||
print_verbose(f"raw model_response: {response}")
|
||||
## RESPONSE OBJECT
|
||||
outputText = "default"
|
||||
if provider == "ai21":
|
||||
outputText = response_body.get('completions')[0].get('data').get('text')
|
||||
else: # amazon titan
|
||||
outputText = response_body.get('results')[0].get('outputText')
|
||||
if "error" in outputText:
|
||||
raise BedrockError(
|
||||
message=outputText,
|
||||
status_code=response.status_code,
|
||||
)
|
||||
else:
|
||||
try:
|
||||
model_response["choices"][0]["message"]["content"] = outputText
|
||||
except:
|
||||
raise BedrockError(message=json.dumps(outputText), status_code=response.status_code)
|
||||
|
||||
## CALCULATING USAGE - baseten charges on time, not tokens - have some mapping of cost here.
|
||||
prompt_tokens = len(
|
||||
encoding.encode(prompt)
|
||||
)
|
||||
completion_tokens = len(
|
||||
encoding.encode(model_response["choices"][0]["message"]["content"])
|
||||
## LOGGING
|
||||
logging_obj.post_call(
|
||||
input=prompt,
|
||||
api_key="",
|
||||
original_response=response,
|
||||
additional_args={"complete_input_dict": data},
|
||||
)
|
||||
print_verbose(f"raw model_response: {response}")
|
||||
## RESPONSE OBJECT
|
||||
outputText = "default"
|
||||
if provider == "ai21":
|
||||
outputText = response_body.get('completions')[0].get('data').get('text')
|
||||
else: # amazon titan
|
||||
outputText = response_body.get('results')[0].get('outputText')
|
||||
if "error" in outputText:
|
||||
raise BedrockError(
|
||||
message=outputText,
|
||||
status_code=response.status_code,
|
||||
)
|
||||
else:
|
||||
try:
|
||||
model_response["choices"][0]["message"]["content"] = outputText
|
||||
except:
|
||||
raise BedrockError(message=json.dumps(outputText), status_code=response.status_code)
|
||||
|
||||
model_response["created"] = time.time()
|
||||
model_response["model"] = model
|
||||
model_response["usage"] = {
|
||||
"prompt_tokens": prompt_tokens,
|
||||
"completion_tokens": completion_tokens,
|
||||
"total_tokens": prompt_tokens + completion_tokens,
|
||||
}
|
||||
return model_response
|
||||
## CALCULATING USAGE - baseten charges on time, not tokens - have some mapping of cost here.
|
||||
prompt_tokens = len(
|
||||
encoding.encode(prompt)
|
||||
)
|
||||
completion_tokens = len(
|
||||
encoding.encode(model_response["choices"][0]["message"]["content"])
|
||||
)
|
||||
|
||||
model_response["created"] = time.time()
|
||||
model_response["model"] = model
|
||||
model_response["usage"] = {
|
||||
"prompt_tokens": prompt_tokens,
|
||||
"completion_tokens": completion_tokens,
|
||||
"total_tokens": prompt_tokens + completion_tokens,
|
||||
}
|
||||
return model_response
|
||||
|
||||
def embedding():
|
||||
# logic for parsing in - calling - parsing out model embedding calls
|
||||
|
|
|
|||
|
|
@ -82,8 +82,7 @@ def mock_completion(model: str, messages: List, stream: bool = False, mock_respo
|
|||
response = mock_completion_streaming_obj(model_response, mock_response=mock_response, model=model)
|
||||
return response
|
||||
|
||||
completion_response = "This is a mock request"
|
||||
model_response["choices"][0]["message"]["content"] = completion_response
|
||||
model_response["choices"][0]["message"]["content"] = mock_response
|
||||
model_response["created"] = time.time()
|
||||
model_response["model"] = model
|
||||
return model_response
|
||||
|
|
@ -133,6 +132,7 @@ def completion(
|
|||
# model specific optional params
|
||||
top_k=40,# used by text-bison only
|
||||
task: Optional[str]="text-generation-inference", # used by huggingface inference endpoints
|
||||
return_full_text: bool = False, # used by huggingface TGI
|
||||
remove_input: bool = True, # used by nlp cloud models - prevents input text from being returned as part of output
|
||||
request_timeout=0, # unused var for old version of OpenAI API
|
||||
fallbacks=[],
|
||||
|
|
@ -162,6 +162,7 @@ def completion(
|
|||
): # allow custom provider to be passed in via the model name "azure/chatgpt-test"
|
||||
custom_llm_provider = model.split("/", 1)[0]
|
||||
model = model.split("/", 1)[1]
|
||||
model, custom_llm_provider = get_llm_provider(model=model, custom_llm_provider=custom_llm_provider)
|
||||
# check if user passed in any of the OpenAI optional params
|
||||
optional_params = get_optional_params(
|
||||
functions=functions,
|
||||
|
|
@ -182,7 +183,8 @@ def completion(
|
|||
custom_llm_provider=custom_llm_provider,
|
||||
top_k=top_k,
|
||||
task=task,
|
||||
remove_input=remove_input
|
||||
remove_input=remove_input,
|
||||
return_full_text=return_full_text
|
||||
)
|
||||
# For logging - save the values of the litellm-specific params passed in
|
||||
litellm_params = get_litellm_params(
|
||||
|
|
@ -198,7 +200,6 @@ def completion(
|
|||
completion_call_id=id
|
||||
)
|
||||
logging.update_environment_variables(model=model, user=user, optional_params=optional_params, litellm_params=litellm_params)
|
||||
get_llm_provider(model=model, custom_llm_provider=custom_llm_provider)
|
||||
if custom_llm_provider == "azure":
|
||||
# azure configs
|
||||
api_type = get_secret("AZURE_API_TYPE") or "azure"
|
||||
|
|
@ -244,7 +245,7 @@ def completion(
|
|||
**optional_params,
|
||||
)
|
||||
if "stream" in optional_params and optional_params["stream"] == True:
|
||||
response = CustomStreamWrapper(response, model, logging_obj=logging)
|
||||
response = CustomStreamWrapper(response, model, custom_llm_provider="openai", logging_obj=logging)
|
||||
return response
|
||||
## LOGGING
|
||||
logging.post_call(
|
||||
|
|
@ -280,7 +281,6 @@ def completion(
|
|||
litellm.openai_key or
|
||||
get_secret("OPENAI_API_KEY")
|
||||
)
|
||||
|
||||
## LOGGING
|
||||
logging.pre_call(
|
||||
input=messages,
|
||||
|
|
@ -310,7 +310,7 @@ def completion(
|
|||
raise e
|
||||
|
||||
if "stream" in optional_params and optional_params["stream"] == True:
|
||||
response = CustomStreamWrapper(response, model, logging_obj=logging)
|
||||
response = CustomStreamWrapper(response, model, custom_llm_provider="openai", logging_obj=logging)
|
||||
return response
|
||||
## LOGGING
|
||||
logging.post_call(
|
||||
|
|
@ -374,7 +374,7 @@ def completion(
|
|||
**optional_params
|
||||
)
|
||||
if "stream" in optional_params and optional_params["stream"] == True:
|
||||
response = CustomStreamWrapper(response, model, logging_obj=logging)
|
||||
response = CustomStreamWrapper(response, model, custom_llm_provider="text-completion-openai", logging_obj=logging)
|
||||
return response
|
||||
## LOGGING
|
||||
logging.post_call(
|
||||
|
|
@ -446,7 +446,7 @@ def completion(
|
|||
)
|
||||
if "stream" in optional_params and optional_params["stream"] == True:
|
||||
# don't try to access stream object,
|
||||
response = CustomStreamWrapper(model_response, model, logging_obj=logging)
|
||||
response = CustomStreamWrapper(model_response, model, custom_llm_provider="anthropic", logging_obj=logging)
|
||||
return response
|
||||
response = model_response
|
||||
elif model in litellm.nlp_cloud_models or custom_llm_provider == "nlp_cloud":
|
||||
|
|
@ -493,7 +493,7 @@ def completion(
|
|||
|
||||
if "stream" in optional_params and optional_params["stream"] == True:
|
||||
# don't try to access stream object,
|
||||
response = CustomStreamWrapper(model_response, model, logging_obj=logging)
|
||||
response = CustomStreamWrapper(model_response, model, custom_llm_provider="aleph-alpha", logging_obj=logging)
|
||||
return response
|
||||
response = model_response
|
||||
elif model in litellm.openrouter_models or custom_llm_provider == "openrouter":
|
||||
|
|
@ -570,7 +570,7 @@ def completion(
|
|||
|
||||
if "stream" in optional_params and optional_params["stream"] == True:
|
||||
# don't try to access stream object,
|
||||
response = CustomStreamWrapper(model_response, model, logging_obj=logging)
|
||||
response = CustomStreamWrapper(model_response, model, custom_llm_provider="cohere", logging_obj=logging)
|
||||
return response
|
||||
response = model_response
|
||||
elif (
|
||||
|
|
@ -782,10 +782,12 @@ def completion(
|
|||
litellm_params=litellm_params,
|
||||
logger_fn=logger_fn,
|
||||
encoding=encoding,
|
||||
logging_obj=logging
|
||||
logging_obj=logging,
|
||||
stream=stream,
|
||||
)
|
||||
|
||||
if "stream" in optional_params and optional_params["stream"] == True: ## [BETA]
|
||||
|
||||
if stream == True:
|
||||
# don't try to access stream object,
|
||||
response = CustomStreamWrapper(
|
||||
iter(model_response), model, custom_llm_provider="bedrock", logging_obj=logging
|
||||
|
|
|
|||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
|
|
@ -91,60 +91,22 @@ def test_completion_with_litellm_call_id():
|
|||
except Exception as e:
|
||||
pytest.fail(f"Error occurred: {e}")
|
||||
|
||||
# commenting out as this is a flaky test on circle ci
|
||||
# def test_completion_nlp_cloud():
|
||||
# try:
|
||||
# messages = [
|
||||
# {"role": "system", "content": "You are a helpful assistant."},
|
||||
# {
|
||||
# "role": "user",
|
||||
# "content": "how does a court case get to the Supreme Court?",
|
||||
# },
|
||||
# ]
|
||||
# response = completion(model="dolphin", messages=messages, logger_fn=logger_fn)
|
||||
# print(response)
|
||||
# except Exception as e:
|
||||
# pytest.fail(f"Error occurred: {e}")
|
||||
|
||||
def test_completion_claude_stream():
|
||||
try:
|
||||
messages = [
|
||||
{"role": "system", "content": "You are a helpful assistant."},
|
||||
{
|
||||
"role": "user",
|
||||
"content": "how does a court case get to the Supreme Court?",
|
||||
},
|
||||
]
|
||||
response = completion(model="claude-2", messages=messages, stream=True)
|
||||
# Add any assertions here to check the response
|
||||
for chunk in response:
|
||||
print(chunk["choices"][0]["delta"]) # same as openai format
|
||||
print(chunk["choices"][0]["finish_reason"])
|
||||
print(chunk["choices"][0]["delta"]["content"])
|
||||
except Exception as e:
|
||||
pytest.fail(f"Error occurred: {e}")
|
||||
# test_completion_claude_stream()
|
||||
|
||||
def test_completion_nlp_cloud():
|
||||
try:
|
||||
messages = [
|
||||
{"role": "system", "content": "You are a helpful assistant."},
|
||||
{
|
||||
"role": "user",
|
||||
"content": "how does a court case get to the Supreme Court?",
|
||||
},
|
||||
]
|
||||
response = completion(model="dolphin", messages=messages, logger_fn=logger_fn)
|
||||
print(response)
|
||||
except Exception as e:
|
||||
pytest.fail(f"Error occurred: {e}")
|
||||
|
||||
def test_completion_nlp_cloud_streaming():
|
||||
try:
|
||||
messages = [
|
||||
{"role": "system", "content": "You are a helpful assistant."},
|
||||
{
|
||||
"role": "user",
|
||||
"content": "how does a court case get to the Supreme Court?",
|
||||
},
|
||||
]
|
||||
response = completion(model="dolphin", messages=messages, stream=True, logger_fn=logger_fn)
|
||||
# Add any assertions here to check the response
|
||||
for chunk in response:
|
||||
print(chunk["choices"][0]["delta"]["content"]) # same as openai format
|
||||
print(chunk["choices"][0]["finish_reason"])
|
||||
print(chunk["choices"][0]["delta"]["content"])
|
||||
except Exception as e:
|
||||
pytest.fail(f"Error occurred: {e}")
|
||||
# test_completion_nlp_cloud_streaming()
|
||||
|
||||
# test_completion_nlp_cloud_streaming()
|
||||
# test_completion_nlp_cloud()
|
||||
# def test_completion_hf_api():
|
||||
# try:
|
||||
# user_message = "write some code to find the sum of two numbers"
|
||||
|
|
@ -193,29 +155,6 @@ def test_completion_cohere(): # commenting for now as the cohere endpoint is bei
|
|||
|
||||
# test_completion_cohere()
|
||||
|
||||
def test_completion_cohere_stream():
|
||||
try:
|
||||
messages = [
|
||||
{"role": "system", "content": "You are a helpful assistant."},
|
||||
{
|
||||
"role": "user",
|
||||
"content": "how does a court case get to the Supreme Court?",
|
||||
},
|
||||
]
|
||||
response = completion(
|
||||
model="command-nightly", messages=messages, stream=True, max_tokens=50
|
||||
)
|
||||
# Add any assertions here to check the response
|
||||
for chunk in response:
|
||||
print(chunk["choices"][0]["delta"]) # same as openai format
|
||||
print(chunk["choices"][0]["finish_reason"])
|
||||
print(chunk["choices"][0]["delta"]["content"])
|
||||
except KeyError:
|
||||
pass
|
||||
except Exception as e:
|
||||
pytest.fail(f"Error occurred: {e}")
|
||||
# test_completion_cohere_stream()
|
||||
|
||||
|
||||
def test_completion_openai():
|
||||
try:
|
||||
|
|
@ -350,70 +289,6 @@ def test_completion_openai_with_more_optional_params():
|
|||
pytest.fail(f"Error occurred: {e}")
|
||||
|
||||
|
||||
def test_completion_openai_with_stream():
|
||||
try:
|
||||
response = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
messages=messages,
|
||||
temperature=0.5,
|
||||
top_p=0.1,
|
||||
n=2,
|
||||
max_tokens=150,
|
||||
presence_penalty=0.5,
|
||||
stream=True,
|
||||
frequency_penalty=-0.5,
|
||||
logit_bias={27000: 5},
|
||||
user="ishaan_dev@berri.ai",
|
||||
)
|
||||
# Add any assertions here to check the response
|
||||
print(response)
|
||||
for chunk in response:
|
||||
print(chunk)
|
||||
if chunk["choices"][0]["finish_reason"] == "stop" or chunk["choices"][0]["finish_reason"] == "length":
|
||||
break
|
||||
print(chunk["choices"][0]["finish_reason"])
|
||||
print(chunk["choices"][0]["delta"]["content"])
|
||||
except Exception as e:
|
||||
pytest.fail(f"Error occurred: {e}")
|
||||
# test_completion_openai_with_stream()
|
||||
|
||||
|
||||
def test_completion_openai_with_functions():
|
||||
function1 = [
|
||||
{
|
||||
"name": "get_current_weather",
|
||||
"description": "Get the current weather in a given location",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"location": {
|
||||
"type": "string",
|
||||
"description": "The city and state, e.g. San Francisco, CA",
|
||||
},
|
||||
"unit": {"type": "string", "enum": ["celsius", "fahrenheit"]},
|
||||
},
|
||||
"required": ["location"],
|
||||
},
|
||||
}
|
||||
]
|
||||
try:
|
||||
response = completion(
|
||||
model="gpt-3.5-turbo", messages=messages, functions=function1, stream=True
|
||||
)
|
||||
# Add any assertions here to check the response
|
||||
print(response)
|
||||
for chunk in response:
|
||||
print(chunk)
|
||||
if chunk["choices"][0]["finish_reason"] == "stop":
|
||||
break
|
||||
print(chunk["choices"][0]["finish_reason"])
|
||||
print(chunk["choices"][0]["delta"]["content"])
|
||||
|
||||
except Exception as e:
|
||||
pytest.fail(f"Error occurred: {e}")
|
||||
# test_completion_openai_with_functions()
|
||||
|
||||
|
||||
# def test_completion_openai_azure_with_functions():
|
||||
# function1 = [
|
||||
# {
|
||||
|
|
@ -568,20 +443,6 @@ def test_completion_replicate_vicuna():
|
|||
except Exception as e:
|
||||
pytest.fail(f"Error occurred: {e}")
|
||||
|
||||
# test_completion_replicate_vicuna()
|
||||
|
||||
def test_completion_replicate_llama_stream():
|
||||
model_name = "replicate/llama-2-70b-chat:2c1608e18606fad2812020dc541930f2d0495ce32eee50074220b87300bc16e1"
|
||||
try:
|
||||
response = completion(model=model_name, messages=messages, stream=True)
|
||||
# Add any assertions here to check the response
|
||||
for chunk in response:
|
||||
print(chunk)
|
||||
print(chunk["choices"][0]["delta"]["content"])
|
||||
except Exception as e:
|
||||
pytest.fail(f"Error occurred: {e}")
|
||||
# test_completion_replicate_llama_stream()
|
||||
|
||||
# def test_completion_replicate_stability_stream():
|
||||
# model_name = "stability-ai/stablelm-tuned-alpha-7b:c49dae362cbaecd2ceabb5bd34fdb68413c4ff775111fea065d259d577757beb"
|
||||
# try:
|
||||
|
|
@ -615,6 +476,7 @@ def test_completion_together_ai():
|
|||
print("Cost for completion call together-computer/llama-2-70b: ", f"${float(cost):.10f}")
|
||||
except Exception as e:
|
||||
pytest.fail(f"Error occurred: {e}")
|
||||
|
||||
# test_completion_together_ai()
|
||||
# def test_customprompt_together_ai():
|
||||
# try:
|
||||
|
|
@ -650,7 +512,8 @@ def test_completion_bedrock_titan():
|
|||
model="bedrock/amazon.titan-tg1-large",
|
||||
messages=messages,
|
||||
temperature=0.2,
|
||||
max_tokens=20,
|
||||
max_tokens=200,
|
||||
top_p=0.8,
|
||||
logger_fn=logger_fn
|
||||
)
|
||||
# Add any assertions here to check the response
|
||||
|
|
@ -662,21 +525,20 @@ def test_completion_bedrock_titan():
|
|||
|
||||
def test_completion_bedrock_ai21():
|
||||
try:
|
||||
litellm.set_verbose = False
|
||||
response = completion(
|
||||
model="bedrock/ai21.j2-mid",
|
||||
messages=messages,
|
||||
temperature=0.2,
|
||||
max_tokens=20,
|
||||
logger_fn=logger_fn
|
||||
top_p=0.2,
|
||||
max_tokens=20
|
||||
)
|
||||
# Add any assertions here to check the response
|
||||
print(response)
|
||||
except Exception as e:
|
||||
pytest.fail(f"Error occurred: {e}")
|
||||
# test_completion_bedrock_ai21()
|
||||
|
||||
|
||||
# test_completion_sagemaker()
|
||||
######## Test VLLM ########
|
||||
# def test_completion_vllm():
|
||||
# try:
|
||||
|
|
@ -761,7 +623,7 @@ def test_completion_bedrock_ai21():
|
|||
# # pass
|
||||
# except Exception as e:
|
||||
# pytest.fail(f"Error occurred: {e}")
|
||||
# test_vertex_ai_stream()
|
||||
# test_vertex_ai_stream()
|
||||
|
||||
|
||||
def test_completion_with_fallbacks():
|
||||
|
|
|
|||
|
|
@ -24,6 +24,171 @@ def logger_fn(model_call_object: dict):
|
|||
user_message = "Hello, how are you?"
|
||||
messages = [{"content": user_message, "role": "user"}]
|
||||
|
||||
|
||||
first_openai_chunk_example = {
|
||||
"id": "chatcmpl-7zSKLBVXnX9dwgRuDYVqVVDsgh2yp",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 1694881253,
|
||||
"model": "gpt-4-0613",
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"delta": {
|
||||
"role": "assistant",
|
||||
"content": ""
|
||||
},
|
||||
"finish_reason": None # it's null
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
def validate_first_format(chunk):
|
||||
# write a test to make sure chunk follows the same format as first_openai_chunk_example
|
||||
assert isinstance(chunk, dict), "Chunk should be a dictionary."
|
||||
assert "id" in chunk, "Chunk should have an 'id'."
|
||||
assert isinstance(chunk['id'], str), "'id' should be a string."
|
||||
|
||||
assert "object" in chunk, "Chunk should have an 'object'."
|
||||
assert isinstance(chunk['object'], str), "'object' should be a string."
|
||||
|
||||
assert "created" in chunk, "Chunk should have a 'created'."
|
||||
assert isinstance(chunk['created'], int), "'created' should be an integer."
|
||||
|
||||
assert "model" in chunk, "Chunk should have a 'model'."
|
||||
assert isinstance(chunk['model'], str), "'model' should be a string."
|
||||
|
||||
assert "choices" in chunk, "Chunk should have 'choices'."
|
||||
assert isinstance(chunk['choices'], list), "'choices' should be a list."
|
||||
|
||||
for choice in chunk['choices']:
|
||||
assert isinstance(choice, dict), "Each choice should be a dictionary."
|
||||
|
||||
assert "index" in choice, "Each choice should have 'index'."
|
||||
assert isinstance(choice['index'], int), "'index' should be an integer."
|
||||
|
||||
assert "delta" in choice, "Each choice should have 'delta'."
|
||||
assert isinstance(choice['delta'], dict), "'delta' should be a dictionary."
|
||||
|
||||
assert "role" in choice['delta'], "'delta' should have a 'role'."
|
||||
assert isinstance(choice['delta']['role'], str), "'role' should be a string."
|
||||
|
||||
assert "content" in choice['delta'], "'delta' should have 'content'."
|
||||
assert isinstance(choice['delta']['content'], str), "'content' should be a string."
|
||||
|
||||
assert "finish_reason" in choice, "Each choice should have 'finish_reason'."
|
||||
assert (choice['finish_reason'] is None) or isinstance(choice['finish_reason'], str), "'finish_reason' should be None or a string."
|
||||
|
||||
second_openai_chunk_example = {
|
||||
"id": "chatcmpl-7zSKLBVXnX9dwgRuDYVqVVDsgh2yp",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 1694881253,
|
||||
"model": "gpt-4-0613",
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"delta": {
|
||||
"content": "Hello"
|
||||
},
|
||||
"finish_reason": None # it's null
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
def validate_second_format(chunk):
|
||||
assert isinstance(chunk, dict), "Chunk should be a dictionary."
|
||||
assert "id" in chunk, "Chunk should have an 'id'."
|
||||
assert isinstance(chunk['id'], str), "'id' should be a string."
|
||||
|
||||
assert "object" in chunk, "Chunk should have an 'object'."
|
||||
assert isinstance(chunk['object'], str), "'object' should be a string."
|
||||
|
||||
assert "created" in chunk, "Chunk should have a 'created'."
|
||||
assert isinstance(chunk['created'], int), "'created' should be an integer."
|
||||
|
||||
assert "model" in chunk, "Chunk should have a 'model'."
|
||||
assert isinstance(chunk['model'], str), "'model' should be a string."
|
||||
|
||||
assert "choices" in chunk, "Chunk should have 'choices'."
|
||||
assert isinstance(chunk['choices'], list), "'choices' should be a list."
|
||||
|
||||
for choice in chunk['choices']:
|
||||
assert isinstance(choice, dict), "Each choice should be a dictionary."
|
||||
|
||||
assert "index" in choice, "Each choice should have 'index'."
|
||||
assert isinstance(choice['index'], int), "'index' should be an integer."
|
||||
|
||||
assert "delta" in choice, "Each choice should have 'delta'."
|
||||
assert isinstance(choice['delta'], dict), "'delta' should be a dictionary."
|
||||
|
||||
assert "content" in choice['delta'], "'delta' should have 'content'."
|
||||
assert isinstance(choice['delta']['content'], str), "'content' should be a string."
|
||||
|
||||
assert "finish_reason" in choice, "Each choice should have 'finish_reason'."
|
||||
assert (choice['finish_reason'] is None) or isinstance(choice['finish_reason'], str), "'finish_reason' should be None or a string."
|
||||
|
||||
last_openai_chunk_example = {
|
||||
"id": "chatcmpl-7zSKLBVXnX9dwgRuDYVqVVDsgh2yp",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 1694881253,
|
||||
"model": "gpt-4-0613",
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"delta": {},
|
||||
"finish_reason": "stop"
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
def validate_last_format(chunk):
|
||||
assert isinstance(chunk, dict), "Chunk should be a dictionary."
|
||||
assert "id" in chunk, "Chunk should have an 'id'."
|
||||
assert isinstance(chunk['id'], str), "'id' should be a string."
|
||||
|
||||
assert "object" in chunk, "Chunk should have an 'object'."
|
||||
assert isinstance(chunk['object'], str), "'object' should be a string."
|
||||
|
||||
assert "created" in chunk, "Chunk should have a 'created'."
|
||||
assert isinstance(chunk['created'], int), "'created' should be an integer."
|
||||
|
||||
assert "model" in chunk, "Chunk should have a 'model'."
|
||||
assert isinstance(chunk['model'], str), "'model' should be a string."
|
||||
|
||||
assert "choices" in chunk, "Chunk should have 'choices'."
|
||||
assert isinstance(chunk['choices'], list), "'choices' should be a list."
|
||||
|
||||
for choice in chunk['choices']:
|
||||
assert isinstance(choice, dict), "Each choice should be a dictionary."
|
||||
|
||||
assert "index" in choice, "Each choice should have 'index'."
|
||||
assert isinstance(choice['index'], int), "'index' should be an integer."
|
||||
|
||||
assert "delta" in choice, "Each choice should have 'delta'."
|
||||
assert isinstance(choice['delta'], dict), "'delta' should be a dictionary."
|
||||
|
||||
assert "finish_reason" in choice, "Each choice should have 'finish_reason'."
|
||||
assert isinstance(choice['finish_reason'], str), "'finish_reason' should be a string."
|
||||
|
||||
def streaming_format_tests(idx, chunk):
|
||||
extracted_chunk = ""
|
||||
finished = False
|
||||
print(f"chunk: {chunk}")
|
||||
if idx == 0: # ensure role assistant is set
|
||||
validate_first_format(chunk=chunk)
|
||||
role = chunk["choices"][0]["delta"]["role"]
|
||||
assert role == "assistant"
|
||||
elif idx == 1: # second chunk
|
||||
validate_second_format(chunk=chunk)
|
||||
if idx != 0: # ensure no role
|
||||
if "role" in chunk["choices"][0]["delta"]:
|
||||
raise Exception("role should not exist after first chunk")
|
||||
if chunk["choices"][0]["finish_reason"]: # ensure finish reason is only in last chunk
|
||||
validate_last_format(chunk=chunk)
|
||||
finished = True
|
||||
if "content" in chunk["choices"][0]["delta"]:
|
||||
extracted_chunk = chunk["choices"][0]["delta"]["content"]
|
||||
return extracted_chunk, finished
|
||||
|
||||
def test_completion_cohere_stream():
|
||||
try:
|
||||
messages = [
|
||||
|
|
@ -38,36 +203,68 @@ def test_completion_cohere_stream():
|
|||
)
|
||||
complete_response = ""
|
||||
# Add any assertions here to check the response
|
||||
for chunk in response:
|
||||
print(f"chunk: {chunk}")
|
||||
complete_response += chunk["choices"][0]["delta"]["content"]
|
||||
if complete_response == "":
|
||||
for idx, chunk in enumerate(response):
|
||||
chunk, finished = streaming_format_tests(idx, chunk)
|
||||
if finished:
|
||||
break
|
||||
complete_response += chunk
|
||||
if complete_response.strip() == "":
|
||||
raise Exception("Empty response received")
|
||||
print(f"completion_response: {complete_response}")
|
||||
except KeyError as e:
|
||||
pass
|
||||
except Exception as e:
|
||||
pytest.fail(f"Error occurred: {e}")
|
||||
|
||||
# test on baseten completion call
|
||||
# try:
|
||||
# response = completion(
|
||||
# model="baseten/RqgAEn0", messages=messages, logger_fn=logger_fn
|
||||
# )
|
||||
# print(f"response: {response}")
|
||||
# complete_response = ""
|
||||
# start_time = time.time()
|
||||
# for chunk in response:
|
||||
# chunk_time = time.time()
|
||||
# print(f"time since initial request: {chunk_time - start_time:.5f}")
|
||||
# print(chunk["choices"][0]["delta"])
|
||||
# complete_response += chunk["choices"][0]["delta"]["content"]
|
||||
# if complete_response == "":
|
||||
# raise Exception("Empty response received")
|
||||
# print(f"complete response: {complete_response}")
|
||||
# except:
|
||||
# print(f"error occurred: {traceback.format_exc()}")
|
||||
# pass
|
||||
|
||||
def test_completion_claude_stream():
|
||||
try:
|
||||
messages = [
|
||||
{"role": "system", "content": "You are a helpful assistant."},
|
||||
{
|
||||
"role": "user",
|
||||
"content": "how does a court case get to the Supreme Court?",
|
||||
},
|
||||
]
|
||||
response = completion(
|
||||
model="claude-instant-1", messages=messages, stream=True, max_tokens=50
|
||||
)
|
||||
complete_response = ""
|
||||
# Add any assertions here to check the response
|
||||
for idx, chunk in enumerate(response):
|
||||
chunk, finished = streaming_format_tests(idx, chunk)
|
||||
if finished:
|
||||
break
|
||||
complete_response += chunk
|
||||
if complete_response.strip() == "":
|
||||
raise Exception("Empty response received")
|
||||
print(f"completion_response: {complete_response}")
|
||||
except Exception as e:
|
||||
pytest.fail(f"Error occurred: {e}")
|
||||
# test_completion_claude_stream()
|
||||
|
||||
def test_completion_bedrock_ai21_stream():
|
||||
try:
|
||||
litellm.set_verbose = False
|
||||
response = completion(
|
||||
model="bedrock/amazon.titan-tg1-large",
|
||||
messages=[{"role": "user", "content": "Be as verbose as possible and give as many details as possible, how does a court case get to the Supreme Court?"}],
|
||||
temperature=1,
|
||||
max_tokens=4096,
|
||||
stream=True,
|
||||
)
|
||||
complete_response = ""
|
||||
# Add any assertions here to check the response
|
||||
print(response)
|
||||
for idx, chunk in enumerate(response):
|
||||
chunk, finished = streaming_format_tests(idx, chunk)
|
||||
if finished:
|
||||
break
|
||||
complete_response += chunk
|
||||
if complete_response.strip() == "":
|
||||
raise Exception("Empty response received")
|
||||
except Exception as e:
|
||||
pytest.fail(f"Error occurred: {e}")
|
||||
|
||||
|
||||
# test_completion_cohere_stream()
|
||||
|
||||
# test on openai completion call
|
||||
def test_openai_text_completion_call():
|
||||
|
|
@ -77,16 +274,15 @@ def test_openai_text_completion_call():
|
|||
)
|
||||
complete_response = ""
|
||||
start_time = time.time()
|
||||
for chunk in response:
|
||||
chunk_time = time.time()
|
||||
print(f"chunk: {chunk}")
|
||||
if "content" in chunk["choices"][0]["delta"]:
|
||||
complete_response += chunk["choices"][0]["delta"]["content"]
|
||||
if complete_response == "":
|
||||
for idx, chunk in enumerate(response):
|
||||
chunk, finished = streaming_format_tests(idx, chunk)
|
||||
if finished:
|
||||
break
|
||||
complete_response += chunk
|
||||
if complete_response.strip() == "":
|
||||
raise Exception("Empty response received")
|
||||
except:
|
||||
print(f"error occurred: {traceback.format_exc()}")
|
||||
pass
|
||||
pytest.fail(f"error occurred: {traceback.format_exc()}")
|
||||
|
||||
# # test on ai21 completion call
|
||||
def ai21_completion_call():
|
||||
|
|
@ -97,18 +293,18 @@ def ai21_completion_call():
|
|||
print(f"response: {response}")
|
||||
complete_response = ""
|
||||
start_time = time.time()
|
||||
for chunk in response:
|
||||
chunk_time = time.time()
|
||||
print(f"time since initial request: {chunk_time - start_time:.5f}")
|
||||
print(chunk)
|
||||
if "content" in chunk["choices"][0]["delta"]:
|
||||
complete_response += chunk["choices"][0]["delta"]["content"]
|
||||
if complete_response == "":
|
||||
for idx, chunk in enumerate(response):
|
||||
chunk, finished = streaming_format_tests(idx, chunk)
|
||||
if finished:
|
||||
break
|
||||
complete_response += chunk
|
||||
if complete_response.strip() == "":
|
||||
raise Exception("Empty response received")
|
||||
print(f"completion_response: {complete_response}")
|
||||
except:
|
||||
print(f"error occurred: {traceback.format_exc()}")
|
||||
pass
|
||||
pytest.fail(f"error occurred: {traceback.format_exc()}")
|
||||
|
||||
# ai21_completion_call()
|
||||
# test on openai completion call
|
||||
def test_openai_chat_completion_call():
|
||||
try:
|
||||
|
|
@ -117,107 +313,20 @@ def test_openai_chat_completion_call():
|
|||
)
|
||||
complete_response = ""
|
||||
start_time = time.time()
|
||||
for chunk in response:
|
||||
print(chunk)
|
||||
if chunk["choices"][0]["finish_reason"]:
|
||||
for idx, chunk in enumerate(response):
|
||||
chunk, finished = streaming_format_tests(idx, chunk)
|
||||
if finished:
|
||||
break
|
||||
# if chunk["choices"][0]["delta"]["role"] != "assistant":
|
||||
# raise Exception("invalid role")
|
||||
if "content" in chunk["choices"][0]["delta"]:
|
||||
complete_response += chunk["choices"][0]["delta"]["content"]
|
||||
complete_response += chunk
|
||||
# print(f'complete_chunk: {complete_response}')
|
||||
if complete_response.strip() == "":
|
||||
raise Exception("Empty response received")
|
||||
print(f"complete response: {complete_response}")
|
||||
except:
|
||||
print(f"error occurred: {traceback.format_exc()}")
|
||||
pass
|
||||
|
||||
test_openai_chat_completion_call()
|
||||
async def completion_call():
|
||||
try:
|
||||
response = completion(
|
||||
model="gpt-3.5-turbo", messages=messages, stream=True, logger_fn=logger_fn
|
||||
)
|
||||
print(f"response: {response}")
|
||||
complete_response = ""
|
||||
start_time = time.time()
|
||||
# Change for loop to async for loop
|
||||
async for chunk in response:
|
||||
chunk_time = time.time()
|
||||
print(f"time since initial request: {chunk_time - start_time:.5f}")
|
||||
print(chunk["choices"][0]["delta"])
|
||||
if "content" in chunk["choices"][0]["delta"]:
|
||||
complete_response += chunk["choices"][0]["delta"]["content"]
|
||||
if complete_response == "":
|
||||
raise Exception("Empty response received")
|
||||
except:
|
||||
print(f"error occurred: {traceback.format_exc()}")
|
||||
pass
|
||||
|
||||
# asyncio.run(completion_call())
|
||||
|
||||
# # test on azure completion call
|
||||
# try:
|
||||
# response = completion(
|
||||
# model="azure/chatgpt-test", messages=messages, stream=True, logger_fn=logger_fn
|
||||
# )
|
||||
# response = ""
|
||||
# start_time = time.time()
|
||||
# for chunk in response:
|
||||
# chunk_time = time.time()
|
||||
# print(f"time since initial request: {chunk_time - start_time:.2f}")
|
||||
# print(chunk["choices"][0]["delta"])
|
||||
# response += chunk["choices"][0]["delta"]
|
||||
# if response == "":
|
||||
# raise Exception("Empty response received")
|
||||
# except:
|
||||
# print(f"error occurred: {traceback.format_exc()}")
|
||||
# pass
|
||||
|
||||
|
||||
# # test on huggingface completion call
|
||||
# try:
|
||||
# start_time = time.time()
|
||||
# response = completion(
|
||||
# model="gpt-3.5-turbo", messages=messages, stream=True, logger_fn=logger_fn
|
||||
# )
|
||||
# complete_response = ""
|
||||
# for chunk in response:
|
||||
# chunk_time = time.time()
|
||||
# print(f"time since initial request: {chunk_time - start_time:.2f}")
|
||||
# print(chunk["choices"][0]["delta"])
|
||||
# complete_response += chunk["choices"][0]["delta"]["content"] if len(chunk["choices"][0]["delta"].keys()) > 0 else ""
|
||||
# if complete_response == "":
|
||||
# raise Exception("Empty response received")
|
||||
# except:
|
||||
# print(f"error occurred: {traceback.format_exc()}")
|
||||
# pass
|
||||
|
||||
# test on together ai completion call - replit-code-3b
|
||||
def test_together_ai_completion_call_replit():
|
||||
try:
|
||||
start_time = time.time()
|
||||
response = completion(
|
||||
model="Replit-Code-3B", messages=messages, logger_fn=logger_fn, stream=True
|
||||
)
|
||||
complete_response = ""
|
||||
print(f"returned response object: {response}")
|
||||
for chunk in response:
|
||||
chunk_time = time.time()
|
||||
print(f"time since initial request: {chunk_time - start_time:.2f}")
|
||||
print(chunk["choices"][0]["delta"])
|
||||
complete_response += (
|
||||
chunk["choices"][0]["delta"]["content"]
|
||||
if len(chunk["choices"][0]["delta"].keys()) > 0
|
||||
else ""
|
||||
)
|
||||
if complete_response == "":
|
||||
raise Exception("Empty response received")
|
||||
except KeyError as e:
|
||||
pass
|
||||
except:
|
||||
print(f"error occurred: {traceback.format_exc()}")
|
||||
pass
|
||||
# test_openai_chat_completion_call()
|
||||
|
||||
# # test on together ai completion call - starcoder
|
||||
def test_together_ai_completion_call_starcoder():
|
||||
|
|
@ -231,50 +340,54 @@ def test_together_ai_completion_call_starcoder():
|
|||
)
|
||||
complete_response = ""
|
||||
print(f"returned response object: {response}")
|
||||
for chunk in response:
|
||||
chunk_time = time.time()
|
||||
complete_response += (
|
||||
chunk["choices"][0]["delta"]["content"]
|
||||
if len(chunk["choices"][0]["delta"].keys()) > 0
|
||||
else ""
|
||||
)
|
||||
if len(complete_response) > 0:
|
||||
print(complete_response)
|
||||
for idx, chunk in enumerate(response):
|
||||
chunk, finished = streaming_format_tests(idx, chunk)
|
||||
if finished:
|
||||
break
|
||||
complete_response += chunk
|
||||
if complete_response == "":
|
||||
raise Exception("Empty response received")
|
||||
except KeyError as e:
|
||||
pass
|
||||
print(f"complete response: {complete_response}")
|
||||
except:
|
||||
print(f"error occurred: {traceback.format_exc()}")
|
||||
pass
|
||||
|
||||
# test on aleph alpha completion call - commented out as it's expensive to run this on circle ci for every build
|
||||
# def test_aleph_alpha_call():
|
||||
# try:
|
||||
# start_time = time.time()
|
||||
# response = completion(
|
||||
# model="luminous-base",
|
||||
# messages=messages,
|
||||
# logger_fn=logger_fn,
|
||||
# stream=True,
|
||||
# )
|
||||
# complete_response = ""
|
||||
# print(f"returned response object: {response}")
|
||||
# for chunk in response:
|
||||
# chunk_time = time.time()
|
||||
# complete_response += (
|
||||
# chunk["choices"][0]["delta"]["content"]
|
||||
# if len(chunk["choices"][0]["delta"].keys()) > 0
|
||||
# else ""
|
||||
# )
|
||||
# if len(complete_response) > 0:
|
||||
# print(complete_response)
|
||||
# if complete_response == "":
|
||||
# raise Exception("Empty response received")
|
||||
# except:
|
||||
# print(f"error occurred: {traceback.format_exc()}")
|
||||
# pass
|
||||
#### Test Async streaming
|
||||
#### Test Function calling + streaming ####
|
||||
|
||||
def test_completion_openai_with_functions():
|
||||
function1 = [
|
||||
{
|
||||
"name": "get_current_weather",
|
||||
"description": "Get the current weather in a given location",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"location": {
|
||||
"type": "string",
|
||||
"description": "The city and state, e.g. San Francisco, CA",
|
||||
},
|
||||
"unit": {"type": "string", "enum": ["celsius", "fahrenheit"]},
|
||||
},
|
||||
"required": ["location"],
|
||||
},
|
||||
}
|
||||
]
|
||||
try:
|
||||
response = completion(
|
||||
model="gpt-3.5-turbo", messages=messages, functions=function1, stream=True
|
||||
)
|
||||
# Add any assertions here to check the response
|
||||
print(response)
|
||||
for chunk in response:
|
||||
print(chunk)
|
||||
if chunk["choices"][0]["finish_reason"] == "stop":
|
||||
break
|
||||
print(chunk["choices"][0]["finish_reason"])
|
||||
print(chunk["choices"][0]["delta"]["content"])
|
||||
except Exception as e:
|
||||
pytest.fail(f"Error occurred: {e}")
|
||||
|
||||
#### Test Async streaming ####
|
||||
|
||||
# # test on ai21 completion call
|
||||
async def ai21_async_completion_call():
|
||||
|
|
@ -286,13 +399,305 @@ async def ai21_async_completion_call():
|
|||
complete_response = ""
|
||||
start_time = time.time()
|
||||
# Change for loop to async for loop
|
||||
idx = 0
|
||||
async for chunk in response:
|
||||
chunk_time = time.time()
|
||||
print(f"time since initial request: {chunk_time - start_time:.5f}")
|
||||
print(chunk["choices"][0]["delta"])
|
||||
complete_response += chunk["choices"][0]["delta"]["content"]
|
||||
if complete_response == "":
|
||||
chunk, finished = streaming_format_tests(idx, chunk)
|
||||
if finished:
|
||||
break
|
||||
complete_response += chunk
|
||||
idx += 1
|
||||
if complete_response.strip() == "":
|
||||
raise Exception("Empty response received")
|
||||
print(f"complete response: {complete_response}")
|
||||
except:
|
||||
print(f"error occurred: {traceback.format_exc()}")
|
||||
pass
|
||||
pass
|
||||
|
||||
# asyncio.run(ai21_async_completion_call())
|
||||
|
||||
async def completion_call():
|
||||
try:
|
||||
response = completion(
|
||||
model="gpt-3.5-turbo", messages=messages, stream=True, logger_fn=logger_fn
|
||||
)
|
||||
print(f"response: {response}")
|
||||
complete_response = ""
|
||||
start_time = time.time()
|
||||
# Change for loop to async for loop
|
||||
idx = 0
|
||||
async for chunk in response:
|
||||
chunk, finished = streaming_format_tests(idx, chunk)
|
||||
if finished:
|
||||
break
|
||||
complete_response += chunk
|
||||
idx += 1
|
||||
if complete_response.strip() == "":
|
||||
raise Exception("Empty response received")
|
||||
print(f"complete response: {complete_response}")
|
||||
except:
|
||||
print(f"error occurred: {traceback.format_exc()}")
|
||||
pass
|
||||
|
||||
# asyncio.run(completion_call())
|
||||
|
||||
#### Test Function Calling + Streaming ####
|
||||
|
||||
final_openai_function_call_example = {
|
||||
"id": "chatcmpl-7zVNA4sXUftpIg6W8WlntCyeBj2JY",
|
||||
"object": "chat.completion",
|
||||
"created": 1694892960,
|
||||
"model": "gpt-3.5-turbo-0613",
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"message": {
|
||||
"role": "assistant",
|
||||
"content": None,
|
||||
"function_call": {
|
||||
"name": "get_current_weather",
|
||||
"arguments": "{\n \"location\": \"Boston, MA\"\n}"
|
||||
}
|
||||
},
|
||||
"finish_reason": "function_call"
|
||||
}
|
||||
],
|
||||
"usage": {
|
||||
"prompt_tokens": 82,
|
||||
"completion_tokens": 18,
|
||||
"total_tokens": 100
|
||||
}
|
||||
}
|
||||
|
||||
function_calling_output_structure = {
|
||||
"id": str,
|
||||
"object": str,
|
||||
"created": int,
|
||||
"model": str,
|
||||
"choices": [
|
||||
{
|
||||
"index": int,
|
||||
"message": {
|
||||
"role": str,
|
||||
"content": (type(None), str),
|
||||
"function_call": {
|
||||
"name": str,
|
||||
"arguments": str
|
||||
}
|
||||
},
|
||||
"finish_reason": str
|
||||
}
|
||||
],
|
||||
"usage": {
|
||||
"prompt_tokens": int,
|
||||
"completion_tokens": int,
|
||||
"total_tokens": int
|
||||
}
|
||||
}
|
||||
|
||||
def validate_final_structure(item, structure=function_calling_output_structure):
|
||||
if isinstance(item, list):
|
||||
if not all(validate_final_structure(i, structure[0]) for i in item):
|
||||
return Exception("Function calling final output doesn't match expected output format")
|
||||
elif isinstance(item, dict):
|
||||
if not all(k in item and validate_final_structure(item[k], v) for k, v in structure.items()):
|
||||
return Exception("Function calling final output doesn't match expected output format")
|
||||
else:
|
||||
if not isinstance(item, structure):
|
||||
return Exception("Function calling final output doesn't match expected output format")
|
||||
return True
|
||||
|
||||
|
||||
first_openai_function_call_example = {
|
||||
"id": "chatcmpl-7zVRoE5HjHYsCMaVSNgOjzdhbS3P0",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 1694893248,
|
||||
"model": "gpt-3.5-turbo-0613",
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"delta": {
|
||||
"role": "assistant",
|
||||
"content": None,
|
||||
"function_call": {
|
||||
"name": "get_current_weather",
|
||||
"arguments": ""
|
||||
}
|
||||
},
|
||||
"finish_reason": None
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
def validate_first_function_call_chunk_structure(item):
|
||||
if not isinstance(item, dict):
|
||||
raise Exception("Incorrect format")
|
||||
|
||||
required_keys = {"id", "object", "created", "model", "choices"}
|
||||
for key in required_keys:
|
||||
if key not in item:
|
||||
raise Exception("Incorrect format")
|
||||
|
||||
if not isinstance(item["choices"], list) or not item["choices"]:
|
||||
raise Exception("Incorrect format")
|
||||
|
||||
required_keys_in_choices_array = {"index", "delta", "finish_reason"}
|
||||
for choice in item["choices"]:
|
||||
if not isinstance(choice, dict):
|
||||
raise Exception("Incorrect format")
|
||||
for key in required_keys_in_choices_array:
|
||||
if key not in choice:
|
||||
raise Exception("Incorrect format")
|
||||
|
||||
if not isinstance(choice["delta"], dict):
|
||||
raise Exception("Incorrect format")
|
||||
|
||||
required_keys_in_delta = {"role", "content", "function_call"}
|
||||
for key in required_keys_in_delta:
|
||||
if key not in choice["delta"]:
|
||||
raise Exception("Incorrect format")
|
||||
|
||||
if not isinstance(choice["delta"]["function_call"], dict):
|
||||
raise Exception("Incorrect format")
|
||||
|
||||
required_keys_in_function_call = {"name", "arguments"}
|
||||
for key in required_keys_in_function_call:
|
||||
if key not in choice["delta"]["function_call"]:
|
||||
raise Exception("Incorrect format")
|
||||
|
||||
return True
|
||||
|
||||
second_function_call_chunk_format = {
|
||||
"id": "chatcmpl-7zVRoE5HjHYsCMaVSNgOjzdhbS3P0",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 1694893248,
|
||||
"model": "gpt-3.5-turbo-0613",
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"delta": {
|
||||
"function_call": {
|
||||
"arguments": "{\n"
|
||||
}
|
||||
},
|
||||
"finish_reason": None
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
|
||||
def validate_second_function_call_chunk_structure(data):
|
||||
if not isinstance(data, dict):
|
||||
raise Exception("Incorrect format")
|
||||
|
||||
required_keys = {"id", "object", "created", "model", "choices"}
|
||||
for key in required_keys:
|
||||
if key not in data:
|
||||
raise Exception("Incorrect format")
|
||||
|
||||
if not isinstance(data["choices"], list) or not data["choices"]:
|
||||
raise Exception("Incorrect format")
|
||||
|
||||
required_keys_in_choices_array = {"index", "delta", "finish_reason"}
|
||||
for choice in data["choices"]:
|
||||
if not isinstance(choice, dict):
|
||||
raise Exception("Incorrect format")
|
||||
for key in required_keys_in_choices_array:
|
||||
if key not in choice:
|
||||
raise Exception("Incorrect format")
|
||||
|
||||
if "function_call" not in choice["delta"] or "arguments" not in choice["delta"]["function_call"]:
|
||||
raise Exception("Incorrect format")
|
||||
|
||||
return True
|
||||
|
||||
|
||||
final_function_call_chunk_example = {
|
||||
"id": "chatcmpl-7zVRoE5HjHYsCMaVSNgOjzdhbS3P0",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 1694893248,
|
||||
"model": "gpt-3.5-turbo-0613",
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"delta": {},
|
||||
"finish_reason": "function_call"
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
|
||||
def validate_final_function_call_chunk_structure(data):
|
||||
if not isinstance(data, dict):
|
||||
raise Exception("Incorrect format")
|
||||
|
||||
required_keys = {"id", "object", "created", "model", "choices"}
|
||||
for key in required_keys:
|
||||
if key not in data:
|
||||
raise Exception("Incorrect format")
|
||||
|
||||
if not isinstance(data["choices"], list) or not data["choices"]:
|
||||
raise Exception("Incorrect format")
|
||||
|
||||
required_keys_in_choices_array = {"index", "delta", "finish_reason"}
|
||||
for choice in data["choices"]:
|
||||
if not isinstance(choice, dict):
|
||||
raise Exception("Incorrect format")
|
||||
for key in required_keys_in_choices_array:
|
||||
if key not in choice:
|
||||
raise Exception("Incorrect format")
|
||||
|
||||
return True
|
||||
|
||||
def streaming_and_function_calling_format_tests(idx, chunk):
|
||||
extracted_chunk = ""
|
||||
finished = False
|
||||
print(f"idx: {idx}")
|
||||
print(f"chunk: {chunk}")
|
||||
decision = False
|
||||
if idx == 0: # ensure role assistant is set
|
||||
decision = validate_first_function_call_chunk_structure(chunk)
|
||||
role = chunk["choices"][0]["delta"]["role"]
|
||||
assert role == "assistant"
|
||||
elif idx != 0: # second chunk
|
||||
try:
|
||||
decision = validate_second_function_call_chunk_structure(data=chunk)
|
||||
except: # check if it's the last chunk (returns an empty delta {} )
|
||||
decision = validate_final_function_call_chunk_structure(data=chunk)
|
||||
finished = True
|
||||
if "content" in chunk["choices"][0]["delta"]:
|
||||
extracted_chunk = chunk["choices"][0]["delta"]["content"]
|
||||
if decision == False:
|
||||
raise Exception("incorrect format")
|
||||
return extracted_chunk, finished
|
||||
|
||||
def test_openai_streaming_and_function_calling():
|
||||
function1 = [
|
||||
{
|
||||
"name": "get_current_weather",
|
||||
"description": "Get the current weather in a given location",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"location": {
|
||||
"type": "string",
|
||||
"description": "The city and state, e.g. San Francisco, CA",
|
||||
},
|
||||
"unit": {"type": "string", "enum": ["celsius", "fahrenheit"]},
|
||||
},
|
||||
"required": ["location"],
|
||||
},
|
||||
}
|
||||
]
|
||||
messages=[{"role": "user", "content": "What is the weather like in Boston?"}]
|
||||
try:
|
||||
response = completion(
|
||||
model="gpt-3.5-turbo", functions=function1, messages=messages, stream=True
|
||||
)
|
||||
# Add any assertions here to check the response
|
||||
for idx, chunk in enumerate(response):
|
||||
streaming_and_function_calling_format_tests(idx=idx, chunk=chunk)
|
||||
except Exception as e:
|
||||
pytest.fail(f"Error occurred: {e}")
|
||||
raise e
|
||||
|
||||
# test_openai_streaming_and_function_calling()
|
||||
|
|
|
|||
202
litellm/utils.py
202
litellm/utils.py
|
|
@ -80,6 +80,8 @@ last_fetched_at_keys = None
|
|||
# 'usage': {'prompt_tokens': 18, 'completion_tokens': 23, 'total_tokens': 41}
|
||||
# }
|
||||
|
||||
def _generate_id(): # private helper function
|
||||
return 'chatcmpl-' + str(uuid.uuid4())
|
||||
|
||||
class Message(OpenAIObject):
|
||||
def __init__(self, content="default", role="assistant", logprobs=None, **params):
|
||||
|
|
@ -89,9 +91,9 @@ class Message(OpenAIObject):
|
|||
self.logprobs = logprobs
|
||||
|
||||
class Delta(OpenAIObject):
|
||||
def __init__(self, content="<special_litellm_token>", logprobs=None, role=None, **params):
|
||||
def __init__(self, content=None, logprobs=None, role=None, **params):
|
||||
super(Delta, self).__init__(**params)
|
||||
if content != "<special_litellm_token>":
|
||||
if content is not None:
|
||||
self.content = content
|
||||
if role:
|
||||
self.role = role
|
||||
|
|
@ -105,20 +107,34 @@ class Choices(OpenAIObject):
|
|||
self.message = message
|
||||
|
||||
class StreamingChoices(OpenAIObject):
|
||||
def __init__(self, finish_reason=None, index=0, delta=Delta(), **params):
|
||||
def __init__(self, finish_reason=None, index=0, delta: Optional[Delta]=None, **params):
|
||||
super(StreamingChoices, self).__init__(**params)
|
||||
self.finish_reason = finish_reason
|
||||
self.index = index
|
||||
self.delta = delta
|
||||
if delta:
|
||||
self.delta = delta
|
||||
else:
|
||||
self.delta = Delta()
|
||||
|
||||
class ModelResponse(OpenAIObject):
|
||||
def __init__(self, choices=None, created=None, model=None, usage=None, stream=False, **params):
|
||||
super(ModelResponse, self).__init__(**params)
|
||||
def __init__(self, id=None, choices=None, created=None, model=None, usage=None, stream=False, **params):
|
||||
if stream:
|
||||
self.choices = self.choices = choices if choices else [StreamingChoices()]
|
||||
self.object = "chat.completion.chunk"
|
||||
self.choices = [StreamingChoices()]
|
||||
else:
|
||||
if model in litellm.open_ai_embedding_models:
|
||||
self.object = "embedding"
|
||||
else:
|
||||
self.object = "chat.completion"
|
||||
self.choices = self.choices = choices if choices else [Choices()]
|
||||
self.created = created
|
||||
if id is None:
|
||||
self.id = _generate_id()
|
||||
else:
|
||||
self.id = id
|
||||
if created is None:
|
||||
self.created = int(time.time())
|
||||
else:
|
||||
self.created = created
|
||||
self.model = model
|
||||
self.usage = (
|
||||
usage
|
||||
|
|
@ -129,6 +145,7 @@ class ModelResponse(OpenAIObject):
|
|||
"total_tokens": None,
|
||||
}
|
||||
)
|
||||
super(ModelResponse, self).__init__(**params)
|
||||
|
||||
def to_dict_recursive(self):
|
||||
d = super().to_dict_recursive()
|
||||
|
|
@ -811,6 +828,7 @@ def get_optional_params( # use the openai defaults
|
|||
model=None,
|
||||
custom_llm_provider="",
|
||||
top_k=40,
|
||||
return_full_text=False,
|
||||
task=None
|
||||
):
|
||||
optional_params = {}
|
||||
|
|
@ -868,9 +886,10 @@ def get_optional_params( # use the openai defaults
|
|||
optional_params["max_new_tokens"] = max_tokens
|
||||
if presence_penalty != 0:
|
||||
optional_params["repetition_penalty"] = presence_penalty
|
||||
optional_params["return_full_text"] = return_full_text
|
||||
optional_params["details"] = True
|
||||
optional_params["task"] = task
|
||||
elif custom_llm_provider == "together_ai" or ("togethercomputer" in model):
|
||||
elif custom_llm_provider == "together_ai":
|
||||
if stream:
|
||||
optional_params["stream_tokens"] = stream
|
||||
if temperature != 1:
|
||||
|
|
@ -931,6 +950,30 @@ def get_optional_params( # use the openai defaults
|
|||
optional_params["temperature"] = temperature
|
||||
if top_p != 1:
|
||||
optional_params["top_p"] = top_p
|
||||
elif custom_llm_provider == "bedrock":
|
||||
if "ai21" in model or "anthropic" in model:
|
||||
# params "maxTokens":200,"temperature":0,"topP":250,"stop_sequences":[],
|
||||
# https://us-west-2.console.aws.amazon.com/bedrock/home?region=us-west-2#/providers?model=j2-ultra
|
||||
if max_tokens != float("inf"):
|
||||
optional_params["maxTokens"] = max_tokens
|
||||
if temperature != 1:
|
||||
optional_params["temperature"] = temperature
|
||||
if stop != None:
|
||||
optional_params["stop_sequences"] = stop
|
||||
if top_p != 1:
|
||||
optional_params["topP"] = top_p
|
||||
|
||||
elif "amazon" in model: # amazon titan llms
|
||||
# see https://us-west-2.console.aws.amazon.com/bedrock/home?region=us-west-2#/providers?model=titan-large
|
||||
if max_tokens != float("inf"):
|
||||
optional_params["maxTokenCount"] = max_tokens
|
||||
if temperature != 1:
|
||||
optional_params["temperature"] = temperature
|
||||
if stop != None:
|
||||
optional_params["stopSequences"] = stop
|
||||
if top_p != 1:
|
||||
optional_params["topP"] = top_p
|
||||
|
||||
elif model in litellm.aleph_alpha_models:
|
||||
if max_tokens != float("inf"):
|
||||
optional_params["maximum_tokens"] = max_tokens
|
||||
|
|
@ -1017,8 +1060,10 @@ def get_llm_provider(model: str, custom_llm_provider: Optional[str] = None):
|
|||
|
||||
# check if model in known model provider list
|
||||
## openai - chatcompletion + text completion
|
||||
if model in litellm.open_ai_chat_completion_models or model in litellm.open_ai_text_completion_models:
|
||||
if model in litellm.open_ai_chat_completion_models:
|
||||
custom_llm_provider = "openai"
|
||||
elif model in litellm.open_ai_text_completion_models:
|
||||
custom_llm_provider = "text-completion-openai"
|
||||
## anthropic
|
||||
elif model in litellm.anthropic_models:
|
||||
custom_llm_provider = "anthropic"
|
||||
|
|
@ -1041,7 +1086,7 @@ def get_llm_provider(model: str, custom_llm_provider: Optional[str] = None):
|
|||
elif model in litellm.ai21_models:
|
||||
custom_llm_provider = "ai21"
|
||||
## together_ai
|
||||
elif model in litellm.together_ai_models or "togethercomputer":
|
||||
elif model in litellm.together_ai_models:
|
||||
custom_llm_provider = "together_ai"
|
||||
## aleph_alpha
|
||||
elif model in litellm.aleph_alpha_models:
|
||||
|
|
@ -2335,6 +2380,7 @@ class CustomStreamWrapper:
|
|||
self.custom_llm_provider = custom_llm_provider
|
||||
self.logging_obj = logging_obj
|
||||
self.completion_stream = completion_stream
|
||||
self.sent_first_chunk = False
|
||||
if self.logging_obj:
|
||||
# Log the type of the received item
|
||||
self.logging_obj.post_call(str(type(completion_stream)))
|
||||
|
|
@ -2406,13 +2452,13 @@ class CustomStreamWrapper:
|
|||
chunk = chunk.decode("utf-8")
|
||||
data_json = json.loads(chunk)
|
||||
try:
|
||||
print(f"data json: {data_json}")
|
||||
return data_json["text"]
|
||||
except:
|
||||
raise ValueError(f"Unable to parse response. Original response: {chunk}")
|
||||
|
||||
def handle_openai_text_completion_chunk(self, chunk):
|
||||
try:
|
||||
print(f"chunk: {chunk}")
|
||||
return chunk["choices"][0]["text"]
|
||||
except:
|
||||
raise ValueError(f"Unable to parse response. Original response: {chunk}")
|
||||
|
|
@ -2451,74 +2497,84 @@ class CustomStreamWrapper:
|
|||
traceback.print_exc()
|
||||
return ""
|
||||
|
||||
def handle_bedrock_stream(self):
|
||||
if self.completion_stream:
|
||||
event = next(self.completion_stream)
|
||||
chunk = event.get('chunk')
|
||||
if chunk:
|
||||
chunk_data = json.loads(chunk.get('bytes').decode())
|
||||
return chunk_data['outputText']
|
||||
return ""
|
||||
|
||||
## needs to handle the empty string case (even starting chunk can be an empty string)
|
||||
def __next__(self):
|
||||
model_response = ModelResponse(stream=True, model=self.model)
|
||||
try:
|
||||
# return this for all models
|
||||
completion_obj = {"content": ""} # default to role being assistant
|
||||
if self.model in litellm.anthropic_models:
|
||||
chunk = next(self.completion_stream)
|
||||
completion_obj["content"] = self.handle_anthropic_chunk(chunk)
|
||||
elif self.model == "replicate" or self.custom_llm_provider == "replicate":
|
||||
chunk = next(self.completion_stream)
|
||||
completion_obj["content"] = chunk
|
||||
elif (
|
||||
self.custom_llm_provider and self.custom_llm_provider == "together_ai"
|
||||
) or ("togethercomputer" in self.model):
|
||||
chunk = next(self.completion_stream)
|
||||
text_data = self.handle_together_ai_chunk(chunk)
|
||||
if text_data == "":
|
||||
return self.__next__()
|
||||
completion_obj["content"] = text_data
|
||||
elif self.custom_llm_provider and self.custom_llm_provider == "huggingface":
|
||||
chunk = next(self.completion_stream)
|
||||
completion_obj["content"] = self.handle_huggingface_chunk(chunk)
|
||||
elif self.custom_llm_provider and self.custom_llm_provider == "baseten": # baseten doesn't provide streaming
|
||||
chunk = next(self.completion_stream)
|
||||
completion_obj["content"] = self.handle_baseten_chunk(chunk)
|
||||
elif self.custom_llm_provider and self.custom_llm_provider == "ai21": #ai21 doesn't provide streaming
|
||||
chunk = next(self.completion_stream)
|
||||
completion_obj["content"] = self.handle_ai21_chunk(chunk)
|
||||
elif self.custom_llm_provider and self.custom_llm_provider == "vllm":
|
||||
chunk = next(self.completion_stream)
|
||||
completion_obj["content"] = chunk[0].outputs[0].text
|
||||
elif self.model in litellm.aleph_alpha_models: #aleph alpha doesn't provide streaming
|
||||
chunk = next(self.completion_stream)
|
||||
completion_obj["content"] = self.handle_aleph_alpha_chunk(chunk)
|
||||
elif self.model in litellm.open_ai_text_completion_models:
|
||||
chunk = next(self.completion_stream)
|
||||
completion_obj["content"] = self.handle_openai_text_completion_chunk(chunk)
|
||||
elif self.model in litellm.nlp_cloud_models or self.custom_llm_provider == "nlp_cloud":
|
||||
chunk = next(self.completion_stream)
|
||||
completion_obj["content"] = self.handle_nlp_cloud_chunk(chunk)
|
||||
elif self.model in (litellm.vertex_chat_models + litellm.vertex_code_chat_models + litellm.vertex_text_models + litellm.vertex_code_text_models):
|
||||
chunk = next(self.completion_stream)
|
||||
completion_obj["content"] = str(chunk)
|
||||
elif self.model in litellm.cohere_models or self.custom_llm_provider == "cohere":
|
||||
chunk = next(self.completion_stream)
|
||||
completion_obj["content"] = self.handle_cohere_chunk(chunk)
|
||||
else: # openai chat/azure models
|
||||
chunk = next(self.completion_stream)
|
||||
model_response = chunk
|
||||
while True: # loop until a non-empty string is found
|
||||
# return this for all models
|
||||
completion_obj = {"content": ""}
|
||||
if self.custom_llm_provider and self.custom_llm_provider == "anthropic":
|
||||
chunk = next(self.completion_stream)
|
||||
completion_obj["content"] = self.handle_anthropic_chunk(chunk)
|
||||
elif self.model == "replicate" or self.custom_llm_provider == "replicate":
|
||||
chunk = next(self.completion_stream)
|
||||
completion_obj["content"] = chunk
|
||||
elif (
|
||||
self.custom_llm_provider and self.custom_llm_provider == "together_ai"):
|
||||
chunk = next(self.completion_stream)
|
||||
text_data = self.handle_together_ai_chunk(chunk)
|
||||
if text_data == "":
|
||||
return self.__next__()
|
||||
completion_obj["content"] = text_data
|
||||
elif self.custom_llm_provider and self.custom_llm_provider == "huggingface":
|
||||
chunk = next(self.completion_stream)
|
||||
completion_obj["content"] = self.handle_huggingface_chunk(chunk)
|
||||
elif self.custom_llm_provider and self.custom_llm_provider == "baseten": # baseten doesn't provide streaming
|
||||
chunk = next(self.completion_stream)
|
||||
completion_obj["content"] = self.handle_baseten_chunk(chunk)
|
||||
elif self.custom_llm_provider and self.custom_llm_provider == "ai21": #ai21 doesn't provide streaming
|
||||
chunk = next(self.completion_stream)
|
||||
completion_obj["content"] = self.handle_ai21_chunk(chunk)
|
||||
elif self.custom_llm_provider and self.custom_llm_provider == "vllm":
|
||||
chunk = next(self.completion_stream)
|
||||
completion_obj["content"] = chunk[0].outputs[0].text
|
||||
elif self.custom_llm_provider and self.custom_llm_provider == "aleph-alpha": #aleph alpha doesn't provide streaming
|
||||
chunk = next(self.completion_stream)
|
||||
completion_obj["content"] = self.handle_aleph_alpha_chunk(chunk)
|
||||
elif self.custom_llm_provider and self.custom_llm_provider == "text-completion-openai":
|
||||
chunk = next(self.completion_stream)
|
||||
completion_obj["content"] = self.handle_openai_text_completion_chunk(chunk)
|
||||
elif self.model in litellm.nlp_cloud_models or self.custom_llm_provider == "nlp_cloud":
|
||||
chunk = next(self.completion_stream)
|
||||
completion_obj["content"] = self.handle_nlp_cloud_chunk(chunk)
|
||||
elif self.model in (litellm.vertex_chat_models + litellm.vertex_code_chat_models + litellm.vertex_text_models + litellm.vertex_code_text_models):
|
||||
chunk = next(self.completion_stream)
|
||||
completion_obj["content"] = str(chunk)
|
||||
elif self.custom_llm_provider == "cohere":
|
||||
chunk = next(self.completion_stream)
|
||||
completion_obj["content"] = self.handle_cohere_chunk(chunk)
|
||||
elif self.custom_llm_provider == "bedrock":
|
||||
completion_obj["content"] = self.handle_bedrock_stream()
|
||||
else: # openai chat/azure models
|
||||
chunk = next(self.completion_stream)
|
||||
model_response = chunk
|
||||
# LOGGING
|
||||
threading.Thread(target=self.logging_obj.success_handler, args=(completion_obj,)).start()
|
||||
return model_response
|
||||
|
||||
# LOGGING
|
||||
threading.Thread(target=self.logging_obj.success_handler, args=(completion_obj,)).start()
|
||||
return model_response
|
||||
|
||||
# LOGGING
|
||||
threading.Thread(target=self.logging_obj.success_handler, args=(completion_obj,)).start()
|
||||
model_response = ModelResponse(stream=True)
|
||||
model_response.choices[0].delta = completion_obj
|
||||
model_response.model = self.model
|
||||
|
||||
if model_response.choices[0].delta['content'] == "<special_litellm_token>":
|
||||
model_response.choices[0].delta = {
|
||||
"content": completion_obj["content"],
|
||||
}
|
||||
return model_response
|
||||
model_response.model = self.model
|
||||
if len(completion_obj["content"]) > 0: # cannot set content of an OpenAI Object to be an empty string
|
||||
if self.sent_first_chunk == False:
|
||||
completion_obj["role"] = "assistant"
|
||||
self.sent_first_chunk = True
|
||||
model_response.choices[0].delta = Delta(**completion_obj)
|
||||
return model_response
|
||||
except StopIteration:
|
||||
raise StopIteration
|
||||
except Exception as e:
|
||||
print(e)
|
||||
model_response = ModelResponse(stream=True)
|
||||
traceback.print_exc()
|
||||
model_response.choices[0].finish_reason = "stop"
|
||||
return model_response
|
||||
|
||||
|
|
|
|||
|
|
@ -1,4 +1,19 @@
|
|||
{
|
||||
"gpt-4": {
|
||||
"max_tokens": 8192,
|
||||
"input_cost_per_token": 0.000003,
|
||||
"output_cost_per_token": 0.00006
|
||||
},
|
||||
"gpt-4-0613": {
|
||||
"max_tokens": 8192,
|
||||
"input_cost_per_token": 0.000003,
|
||||
"output_cost_per_token": 0.00006
|
||||
},
|
||||
"gpt-4-32k": {
|
||||
"max_tokens": 32768,
|
||||
"input_cost_per_token": 0.00006,
|
||||
"output_cost_per_token": 0.00012
|
||||
},
|
||||
"gpt-3.5-turbo": {
|
||||
"max_tokens": 4097,
|
||||
"input_cost_per_token": 0.0000015,
|
||||
|
|
@ -24,21 +39,6 @@
|
|||
"input_cost_per_token": 0.000003,
|
||||
"output_cost_per_token": 0.000004
|
||||
},
|
||||
"gpt-4": {
|
||||
"max_tokens": 8192,
|
||||
"input_cost_per_token": 0.000003,
|
||||
"output_cost_per_token": 0.00006
|
||||
},
|
||||
"gpt-4-0613": {
|
||||
"max_tokens": 8192,
|
||||
"input_cost_per_token": 0.000003,
|
||||
"output_cost_per_token": 0.00006
|
||||
},
|
||||
"gpt-4-32k": {
|
||||
"max_tokens": 32768,
|
||||
"input_cost_per_token": 0.00006,
|
||||
"output_cost_per_token": 0.00012
|
||||
},
|
||||
"claude-instant-1": {
|
||||
"max_tokens": 100000,
|
||||
"input_cost_per_token": 0.00000163,
|
||||
|
|
|
|||
BIN
proxy-server/.DS_Store
vendored
BIN
proxy-server/.DS_Store
vendored
Binary file not shown.
|
|
@ -1,6 +1,6 @@
|
|||
[tool.poetry]
|
||||
name = "litellm"
|
||||
version = "0.1.674"
|
||||
version = "0.1.687"
|
||||
description = "Library to easily interface with LLM API providers"
|
||||
authors = ["BerriAI"]
|
||||
license = "MIT License"
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue