mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
docs: replace gpt-3.5-turbo with gpt-4o in example docs
Co-authored-by: Krish Dholakia <krrishdholakia@gmail.com>
This commit is contained in:
parent
8e12f0a277
commit
2a2386071d
13 changed files with 119 additions and 119 deletions
|
|
@ -51,7 +51,7 @@ if not budget_manager.is_valid_user(user):
|
|||
|
||||
# check if a given call can be made
|
||||
if budget_manager.get_current_cost(user=user) <= budget_manager.get_total_budget(user):
|
||||
response = completion(model="gpt-3.5-turbo", messages=[{"role": "user", "content": "Hey, how's it going?"}])
|
||||
response = completion(model="gpt-4o", messages=[{"role": "user", "content": "Hey, how's it going?"}])
|
||||
budget_manager.update_cost(completion_obj=response, user=user)
|
||||
else:
|
||||
response = "Sorry - no budget!"
|
||||
|
|
@ -72,7 +72,7 @@ budget_manager.create_budget(total_budget=10, user=user, duration="daily")
|
|||
|
||||
input_text = "hello world"
|
||||
output_text = "it's a sunny day in san francisco"
|
||||
model = "gpt-3.5-turbo"
|
||||
model = "gpt-4o"
|
||||
|
||||
budget_manager.update_cost(user=user, model=model, input_text=input_text, output_text=output_text) # 👈
|
||||
print(budget_manager.get_current_cost(user))
|
||||
|
|
@ -108,7 +108,7 @@ if not budget_manager.is_valid_user(user):
|
|||
|
||||
# check if a given call can be made
|
||||
if budget_manager.get_current_cost(user=user) <= budget_manager.get_total_budget(user):
|
||||
response = completion(model="gpt-3.5-turbo", messages=[{"role": "user", "content": "Hey, how's it going?"}])
|
||||
response = completion(model="gpt-4o", messages=[{"role": "user", "content": "Hey, how's it going?"}])
|
||||
budget_manager.update_cost(completion_obj=response, user=user)
|
||||
else:
|
||||
response = "Sorry - no budget!"
|
||||
|
|
@ -138,7 +138,7 @@ if not budget_manager.is_valid_user(user):
|
|||
|
||||
# check if a given call can be made
|
||||
if budget_manager.get_current_cost(user=user) <= budget_manager.get_total_budget(user):
|
||||
response = completion(model="gpt-3.5-turbo", messages=[{"role": "user", "content": "Hey, how's it going?"}])
|
||||
response = completion(model="gpt-4o", messages=[{"role": "user", "content": "Hey, how's it going?"}])
|
||||
budget_manager.update_cost(completion_obj=response, user=user)
|
||||
else:
|
||||
response = "Sorry - no budget!"
|
||||
|
|
|
|||
|
|
@ -69,7 +69,7 @@ except openai.APITimeoutError as e:
|
|||
import litellm
|
||||
try:
|
||||
response = litellm.completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
|
|
|
|||
|
|
@ -36,7 +36,7 @@ model_list = [
|
|||
{
|
||||
"model_name": "fake-openai-endpoint",
|
||||
"litellm_params": {
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"api_key": "my-fake-key",
|
||||
"api_base": "http://0.0.0.0:8080",
|
||||
"rpm": 100
|
||||
|
|
@ -45,7 +45,7 @@ model_list = [
|
|||
{
|
||||
"model_name": "fake-openai-endpoint",
|
||||
"litellm_params": {
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"api_key": "my-fake-key",
|
||||
"api_base": "http://0.0.0.0:8081",
|
||||
"rpm": 100
|
||||
|
|
|
|||
|
|
@ -42,7 +42,7 @@ curl http://localhost:4000/chat/completions \
|
|||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{"role": "user", "content": "My credit card is 4111-1111-1111-1111 and my email is john@example.com"}
|
||||
],
|
||||
|
|
@ -68,7 +68,7 @@ client = openai.OpenAI(
|
|||
|
||||
# This request will trigger MCP guardrails
|
||||
response = client.chat.completions.create(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[
|
||||
{"role": "user", "content": "Send an email to 555-123-4567 with my SSN 123-45-6789"}
|
||||
],
|
||||
|
|
|
|||
|
|
@ -18,12 +18,12 @@ When we have breaking changes (i.e. going from 1.x.x to 2.x.x), we will document
|
|||
- *NEW* default exception - `APIConnectionError` (prev. `APIError`)
|
||||
- litellm.get_max_tokens() now returns an int not a dict
|
||||
```python
|
||||
max_tokens = litellm.get_max_tokens("gpt-3.5-turbo") # returns an int not a dict
|
||||
max_tokens = litellm.get_max_tokens("gpt-4o") # returns an int not a dict
|
||||
assert max_tokens==4097
|
||||
```
|
||||
- Streaming - OpenAI Chunks now return `None` for empty stream chunks. This is how to process stream chunks with content
|
||||
```python
|
||||
response = litellm.completion(model="gpt-3.5-turbo", messages=messages, stream=True)
|
||||
response = litellm.completion(model="gpt-4o", messages=messages, stream=True)
|
||||
for part in response:
|
||||
print(part.choices[0].delta.content or "")
|
||||
```
|
||||
|
|
|
|||
|
|
@ -11,9 +11,9 @@ Setup Prompt Injection Detection, Secret Detection on LiteLLM Proxy
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: openai/gpt-3.5-turbo
|
||||
model: openai/gpt-4o
|
||||
api_key: sk-xxxxxxx
|
||||
|
||||
litellm_settings:
|
||||
|
|
@ -56,7 +56,7 @@ curl --location 'http://localhost:4000/chat/completions' \
|
|||
--header 'Authorization: Bearer sk-1234' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -280,7 +280,7 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \
|
|||
--header 'Content-Type: application/json' \
|
||||
--header 'Authorization: Bearer $LITELLM_VIRTUAL_KEY' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
|
|||
|
|
@ -15,7 +15,7 @@ os.environ["COHERE_API_KEY"] = "your-api-key"
|
|||
messages = [{ "content": "Hello, how are you?","role": "user"}]
|
||||
|
||||
# openai call
|
||||
response = completion(model="gpt-3.5-turbo", messages=messages)
|
||||
response = completion(model="gpt-4o", messages=messages)
|
||||
|
||||
# cohere call
|
||||
response = completion("command-nightly", messages)
|
||||
|
|
@ -31,8 +31,8 @@ For a complete list of models/providers that you can call with LiteLLM, [check o
|
|||
|
||||
* OpenAI models - [OpenAI docs](./providers/openai.md)
|
||||
* gpt-4
|
||||
* gpt-3.5-turbo
|
||||
* gpt-3.5-turbo-16k
|
||||
* gpt-4o
|
||||
* gpt-4o-16k
|
||||
* Llama2 models - [TogetherAI docs](./providers/togetherai.md)
|
||||
* togethercomputer/llama-2-70b-chat
|
||||
* togethercomputer/llama-2-70b
|
||||
|
|
|
|||
|
|
@ -520,7 +520,7 @@ assert litellm.supports_reasoning(model="anthropic/claude-3-7-sonnet-20250219")
|
|||
assert litellm.supports_reasoning(model="deepseek/deepseek-chat") == True
|
||||
|
||||
# Example models that do not support reasoning
|
||||
assert litellm.supports_reasoning(model="openai/gpt-3.5-turbo") == False
|
||||
assert litellm.supports_reasoning(model="openai/gpt-4o") == False
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
|
|
|
|||
|
|
@ -34,7 +34,7 @@ Loadbalance across multiple [azure](./providers/azure)/[bedrock](./providers/bed
|
|||
from litellm import Router
|
||||
|
||||
model_list = [{ # list of model deployments
|
||||
"model_name": "gpt-3.5-turbo", # model alias -> loadbalance between models with same `model_name`
|
||||
"model_name": "gpt-4o", # model alias -> loadbalance between models with same `model_name`
|
||||
"litellm_params": { # params for litellm completion/embedding call
|
||||
"model": "azure/chatgpt-v-2", # actual model name
|
||||
"api_key": os.getenv("AZURE_API_KEY"),
|
||||
|
|
@ -42,7 +42,7 @@ model_list = [{ # list of model deployments
|
|||
"api_base": os.getenv("AZURE_API_BASE")
|
||||
}
|
||||
}, {
|
||||
"model_name": "gpt-3.5-turbo",
|
||||
"model_name": "gpt-4o",
|
||||
"litellm_params": { # params for litellm completion/embedding call
|
||||
"model": "azure/chatgpt-functioncalling",
|
||||
"api_key": os.getenv("AZURE_API_KEY"),
|
||||
|
|
@ -50,9 +50,9 @@ model_list = [{ # list of model deployments
|
|||
"api_base": os.getenv("AZURE_API_BASE")
|
||||
}
|
||||
}, {
|
||||
"model_name": "gpt-3.5-turbo",
|
||||
"model_name": "gpt-4o",
|
||||
"litellm_params": { # params for litellm completion/embedding call
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"api_key": os.getenv("OPENAI_API_KEY"),
|
||||
}
|
||||
}, {
|
||||
|
|
@ -76,8 +76,8 @@ model_list = [{ # list of model deployments
|
|||
router = Router(model_list=model_list)
|
||||
|
||||
# openai.ChatCompletion.create replacement
|
||||
# requests with model="gpt-3.5-turbo" will pick a deployment where model_name="gpt-3.5-turbo"
|
||||
response = await router.acompletion(model="gpt-3.5-turbo",
|
||||
# requests with model="gpt-4o" will pick a deployment where model_name="gpt-4o"
|
||||
response = await router.acompletion(model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Hey, how's it going?"}])
|
||||
|
||||
print(response)
|
||||
|
|
@ -101,17 +101,17 @@ See detailed proxy loadbalancing/fallback docs [here](./proxy/reliability.md)
|
|||
1. Setup model_list with multiple deployments
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: azure/<your-deployment-name>
|
||||
api_base: <your-azure-endpoint>
|
||||
api_key: <your-azure-api-key>
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: azure/gpt-turbo-small-ca
|
||||
api_base: https://my-endpoint-canada-berri992.openai.azure.com/
|
||||
api_key: <your-azure-api-key>
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: azure/gpt-turbo-large
|
||||
api_base: https://openai-france-1234.openai.azure.com/
|
||||
|
|
@ -131,7 +131,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
|||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{"role": "user", "content": "Hi there!"}
|
||||
],
|
||||
|
|
@ -174,14 +174,14 @@ You can also set a `weight` param, to specify which model should get picked when
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: azure/chatgpt-v-2
|
||||
api_key: os.environ/AZURE_API_KEY
|
||||
api_version: os.environ/AZURE_API_VERSION
|
||||
api_base: os.environ/AZURE_API_BASE
|
||||
rpm: 900
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: azure/chatgpt-functioncalling
|
||||
api_key: os.environ/AZURE_API_KEY
|
||||
|
|
@ -197,7 +197,7 @@ from litellm import Router
|
|||
import asyncio
|
||||
|
||||
model_list = [{ # list of model deployments
|
||||
"model_name": "gpt-3.5-turbo", # model alias
|
||||
"model_name": "gpt-4o", # model alias
|
||||
"litellm_params": { # params for litellm completion/embedding call
|
||||
"model": "azure/chatgpt-v-2", # actual model name
|
||||
"api_key": os.getenv("AZURE_API_KEY"),
|
||||
|
|
@ -206,7 +206,7 @@ model_list = [{ # list of model deployments
|
|||
"rpm": 900, # requests per minute for this API
|
||||
}
|
||||
}, {
|
||||
"model_name": "gpt-3.5-turbo",
|
||||
"model_name": "gpt-4o",
|
||||
"litellm_params": { # params for litellm completion/embedding call
|
||||
"model": "azure/chatgpt-functioncalling",
|
||||
"api_key": os.getenv("AZURE_API_KEY"),
|
||||
|
|
@ -220,7 +220,7 @@ model_list = [{ # list of model deployments
|
|||
router = Router(model_list=model_list, routing_strategy="simple-shuffle")
|
||||
async def router_acompletion():
|
||||
response = await router.acompletion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Hey, how's it going?"}]
|
||||
)
|
||||
print(response)
|
||||
|
|
@ -236,14 +236,14 @@ asyncio.run(router_acompletion())
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: azure/chatgpt-v-2
|
||||
api_key: os.environ/AZURE_API_KEY
|
||||
api_version: os.environ/AZURE_API_VERSION
|
||||
api_base: os.environ/AZURE_API_BASE
|
||||
weight: 9
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: azure/chatgpt-functioncalling
|
||||
api_key: os.environ/AZURE_API_KEY
|
||||
|
|
@ -259,7 +259,7 @@ from litellm import Router
|
|||
import asyncio
|
||||
|
||||
model_list = [{
|
||||
"model_name": "gpt-3.5-turbo", # model alias
|
||||
"model_name": "gpt-4o", # model alias
|
||||
"litellm_params": {
|
||||
"model": "azure/chatgpt-v-2", # actual model name
|
||||
"api_key": os.getenv("AZURE_API_KEY"),
|
||||
|
|
@ -268,7 +268,7 @@ model_list = [{
|
|||
"weight": 9, # pick this 90% of the time
|
||||
}
|
||||
}, {
|
||||
"model_name": "gpt-3.5-turbo",
|
||||
"model_name": "gpt-4o",
|
||||
"litellm_params": {
|
||||
"model": "azure/chatgpt-functioncalling",
|
||||
"api_key": os.getenv("AZURE_API_KEY"),
|
||||
|
|
@ -282,7 +282,7 @@ model_list = [{
|
|||
router = Router(model_list=model_list, routing_strategy="simple-shuffle")
|
||||
async def router_acompletion():
|
||||
response = await router.acompletion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Hey, how's it going?"}]
|
||||
)
|
||||
print(response)
|
||||
|
|
@ -319,7 +319,7 @@ from litellm import Router
|
|||
|
||||
|
||||
model_list = [{ # list of model deployments
|
||||
"model_name": "gpt-3.5-turbo", # model alias
|
||||
"model_name": "gpt-4o", # model alias
|
||||
"litellm_params": { # params for litellm completion/embedding call
|
||||
"model": "azure/chatgpt-v-2", # actual model name
|
||||
"api_key": os.getenv("AZURE_API_KEY"),
|
||||
|
|
@ -329,7 +329,7 @@ model_list = [{ # list of model deployments
|
|||
"rpm": 10000,
|
||||
},
|
||||
}, {
|
||||
"model_name": "gpt-3.5-turbo",
|
||||
"model_name": "gpt-4o",
|
||||
"litellm_params": { # params for litellm completion/embedding call
|
||||
"model": "azure/chatgpt-functioncalling",
|
||||
"api_key": os.getenv("AZURE_API_KEY"),
|
||||
|
|
@ -339,9 +339,9 @@ model_list = [{ # list of model deployments
|
|||
"rpm": 1000,
|
||||
},
|
||||
}, {
|
||||
"model_name": "gpt-3.5-turbo",
|
||||
"model_name": "gpt-4o",
|
||||
"litellm_params": { # params for litellm completion/embedding call
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"api_key": os.getenv("OPENAI_API_KEY"),
|
||||
"tpm": 100000,
|
||||
"rpm": 1000,
|
||||
|
|
@ -355,7 +355,7 @@ router = Router(model_list=model_list,
|
|||
enable_pre_call_checks=True, # enables router rate limits for concurrent calls
|
||||
)
|
||||
|
||||
response = await router.acompletion(model="gpt-3.5-turbo",
|
||||
response = await router.acompletion(model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Hey, how's it going?"}]
|
||||
|
||||
print(response)
|
||||
|
|
@ -367,7 +367,7 @@ print(response)
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo # model alias
|
||||
- model_name: gpt-4o # model alias
|
||||
litellm_params: # params for litellm completion/embedding call
|
||||
model: azure/chatgpt-v-2 # actual model name
|
||||
api_key: os.environ/AZURE_API_KEY
|
||||
|
|
@ -375,9 +375,9 @@ model_list:
|
|||
api_base: os.environ/AZURE_API_BASE
|
||||
tpm: 100000
|
||||
rpm: 10000
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params: # params for litellm completion/embedding call
|
||||
model: gpt-3.5-turbo
|
||||
model: gpt-4o
|
||||
api_key: os.getenv(OPENAI_API_KEY)
|
||||
tpm: 100000
|
||||
rpm: 1000
|
||||
|
|
@ -406,7 +406,7 @@ curl --location 'http://localhost:4000/v1/chat/completions' \
|
|||
--header 'Content-Type: application/json' \
|
||||
--header 'Authorization: Bearer sk-1234' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [{"role": "user", "content": "Hey, how's it going?"}]
|
||||
}'
|
||||
```
|
||||
|
|
@ -524,7 +524,7 @@ from litellm import Router
|
|||
|
||||
|
||||
model_list = [{ # list of model deployments
|
||||
"model_name": "gpt-3.5-turbo", # model alias
|
||||
"model_name": "gpt-4o", # model alias
|
||||
"litellm_params": { # params for litellm completion/embedding call
|
||||
"model": "azure/chatgpt-v-2", # actual model name
|
||||
"api_key": os.getenv("AZURE_API_KEY"),
|
||||
|
|
@ -534,7 +534,7 @@ model_list = [{ # list of model deployments
|
|||
"tpm": 100000,
|
||||
"rpm": 10000,
|
||||
}, {
|
||||
"model_name": "gpt-3.5-turbo",
|
||||
"model_name": "gpt-4o",
|
||||
"litellm_params": { # params for litellm completion/embedding call
|
||||
"model": "azure/chatgpt-functioncalling",
|
||||
"api_key": os.getenv("AZURE_API_KEY"),
|
||||
|
|
@ -544,9 +544,9 @@ model_list = [{ # list of model deployments
|
|||
"tpm": 100000,
|
||||
"rpm": 1000,
|
||||
}, {
|
||||
"model_name": "gpt-3.5-turbo",
|
||||
"model_name": "gpt-4o",
|
||||
"litellm_params": { # params for litellm completion/embedding call
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"api_key": os.getenv("OPENAI_API_KEY"),
|
||||
},
|
||||
"tpm": 100000,
|
||||
|
|
@ -560,7 +560,7 @@ router = Router(model_list=model_list,
|
|||
enable_pre_call_check=True, # enables router rate limits for concurrent calls
|
||||
)
|
||||
|
||||
response = await router.acompletion(model="gpt-3.5-turbo",
|
||||
response = await router.acompletion(model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Hey, how's it going?"}]
|
||||
|
||||
print(response)
|
||||
|
|
@ -580,7 +580,7 @@ from litellm import Router
|
|||
import asyncio
|
||||
|
||||
model_list = [{ # list of model deployments
|
||||
"model_name": "gpt-3.5-turbo", # model alias
|
||||
"model_name": "gpt-4o", # model alias
|
||||
"litellm_params": { # params for litellm completion/embedding call
|
||||
"model": "azure/chatgpt-v-2", # actual model name
|
||||
"api_key": os.getenv("AZURE_API_KEY"),
|
||||
|
|
@ -588,7 +588,7 @@ model_list = [{ # list of model deployments
|
|||
"api_base": os.getenv("AZURE_API_BASE"),
|
||||
}
|
||||
}, {
|
||||
"model_name": "gpt-3.5-turbo",
|
||||
"model_name": "gpt-4o",
|
||||
"litellm_params": { # params for litellm completion/embedding call
|
||||
"model": "azure/chatgpt-functioncalling",
|
||||
"api_key": os.getenv("AZURE_API_KEY"),
|
||||
|
|
@ -596,9 +596,9 @@ model_list = [{ # list of model deployments
|
|||
"api_base": os.getenv("AZURE_API_BASE"),
|
||||
}
|
||||
}, {
|
||||
"model_name": "gpt-3.5-turbo",
|
||||
"model_name": "gpt-4o",
|
||||
"litellm_params": { # params for litellm completion/embedding call
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"api_key": os.getenv("OPENAI_API_KEY"),
|
||||
}
|
||||
}]
|
||||
|
|
@ -607,7 +607,7 @@ model_list = [{ # list of model deployments
|
|||
router = Router(model_list=model_list, routing_strategy="least-busy")
|
||||
async def router_acompletion():
|
||||
response = await router.acompletion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Hey, how's it going?"}]
|
||||
)
|
||||
print(response)
|
||||
|
|
@ -750,12 +750,12 @@ import asyncio
|
|||
|
||||
model_list = [
|
||||
{
|
||||
"model_name": "gpt-3.5-turbo",
|
||||
"model_name": "gpt-4o",
|
||||
"litellm_params": {"model": "gpt-4"},
|
||||
"model_info": {"id": "openai-gpt-4"},
|
||||
},
|
||||
{
|
||||
"model_name": "gpt-3.5-turbo",
|
||||
"model_name": "gpt-4o",
|
||||
"litellm_params": {"model": "groq/llama3-8b-8192"},
|
||||
"model_info": {"id": "groq-llama"},
|
||||
},
|
||||
|
|
@ -765,7 +765,7 @@ model_list = [
|
|||
router = Router(model_list=model_list, routing_strategy="cost-based-routing")
|
||||
async def router_acompletion():
|
||||
response = await router.acompletion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Hey, how's it going?"}]
|
||||
)
|
||||
print(response)
|
||||
|
|
@ -785,7 +785,7 @@ Set `litellm_params["input_cost_per_token"]` and `litellm_params["output_cost_pe
|
|||
```python
|
||||
model_list = [
|
||||
{
|
||||
"model_name": "gpt-3.5-turbo",
|
||||
"model_name": "gpt-4o",
|
||||
"litellm_params": {
|
||||
"model": "azure/chatgpt-v-2",
|
||||
"input_cost_per_token": 0.00003,
|
||||
|
|
@ -794,7 +794,7 @@ model_list = [
|
|||
"model_info": {"id": "chatgpt-v-experimental"},
|
||||
},
|
||||
{
|
||||
"model_name": "gpt-3.5-turbo",
|
||||
"model_name": "gpt-4o",
|
||||
"litellm_params": {
|
||||
"model": "azure/chatgpt-v-1",
|
||||
"input_cost_per_token": 0.000000001,
|
||||
|
|
@ -803,7 +803,7 @@ model_list = [
|
|||
"model_info": {"id": "chatgpt-v-1"},
|
||||
},
|
||||
{
|
||||
"model_name": "gpt-3.5-turbo",
|
||||
"model_name": "gpt-4o",
|
||||
"litellm_params": {
|
||||
"model": "azure/chatgpt-v-5",
|
||||
"input_cost_per_token": 10,
|
||||
|
|
@ -816,7 +816,7 @@ model_list = [
|
|||
router = Router(model_list=model_list, routing_strategy="cost-based-routing")
|
||||
async def router_acompletion():
|
||||
response = await router.acompletion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Hey, how's it going?"}]
|
||||
)
|
||||
print(response)
|
||||
|
|
@ -932,7 +932,7 @@ model_list = [
|
|||
router = Router(model_list=model_list, routing_strategy="cost-based-routing")
|
||||
|
||||
response = await router.acompletion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Hey, how's it going?"}]
|
||||
)
|
||||
print(response)
|
||||
|
|
@ -1007,7 +1007,7 @@ user_message = "Hello, whats the weather in San Francisco??"
|
|||
messages = [{"content": user_message, "role": "user"}]
|
||||
|
||||
# normal call
|
||||
response = router.completion(model="gpt-3.5-turbo", messages=messages)
|
||||
response = router.completion(model="gpt-4o", messages=messages)
|
||||
|
||||
print(f"response: {response}")
|
||||
```
|
||||
|
|
@ -1190,7 +1190,7 @@ user_message = "Hello, whats the weather in San Francisco??"
|
|||
messages = [{"content": user_message, "role": "user"}]
|
||||
|
||||
# normal call
|
||||
response = router.completion(model="gpt-3.5-turbo", messages=messages)
|
||||
response = router.completion(model="gpt-4o", messages=messages)
|
||||
|
||||
print(f"response: {response}")
|
||||
```
|
||||
|
|
@ -1209,7 +1209,7 @@ user_message = "Hello, whats the weather in San Francisco??"
|
|||
messages = [{"content": user_message, "role": "user"}]
|
||||
|
||||
# normal call
|
||||
response = router.completion(model="gpt-3.5-turbo", messages=messages)
|
||||
response = router.completion(model="gpt-4o", messages=messages)
|
||||
|
||||
print(f"response: {response}")
|
||||
```
|
||||
|
|
@ -1260,7 +1260,7 @@ allowed_fails_policy = AllowedFailsPolicy(
|
|||
router = litellm.Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "gpt-3.5-turbo", # openai model name
|
||||
"model_name": "gpt-4o", # openai model name
|
||||
"litellm_params": { # params for litellm completion/embedding call
|
||||
"model": "azure/chatgpt-v-2",
|
||||
"api_key": os.getenv("AZURE_API_KEY"),
|
||||
|
|
@ -1374,7 +1374,7 @@ For 'eu-region' filtering, Set 'region_name' of deployment.
|
|||
```python
|
||||
model_list = [
|
||||
{
|
||||
"model_name": "gpt-3.5-turbo", # model group name
|
||||
"model_name": "gpt-4o", # model group name
|
||||
"litellm_params": { # params for litellm completion/embedding call
|
||||
"model": "azure/chatgpt-v-2",
|
||||
"api_key": os.getenv("AZURE_API_KEY"),
|
||||
|
|
@ -1385,9 +1385,9 @@ model_list = [
|
|||
},
|
||||
},
|
||||
{
|
||||
"model_name": "gpt-3.5-turbo", # model group name
|
||||
"model_name": "gpt-4o", # model group name
|
||||
"litellm_params": { # params for litellm completion/embedding call
|
||||
"model": "gpt-3.5-turbo-1106",
|
||||
"model": "gpt-4o-1106",
|
||||
"api_key": os.getenv("OPENAI_API_KEY"),
|
||||
},
|
||||
},
|
||||
|
|
@ -1413,7 +1413,7 @@ router = Router(model_list=model_list, enable_pre_call_checks=True)
|
|||
|
||||
```python
|
||||
"""
|
||||
- Give a gpt-3.5-turbo model group with different context windows (4k vs. 16k)
|
||||
- Give a gpt-4o model group with different context windows (4k vs. 16k)
|
||||
- Send a 5k prompt
|
||||
- Assert it works
|
||||
"""
|
||||
|
|
@ -1422,7 +1422,7 @@ import os
|
|||
|
||||
model_list = [
|
||||
{
|
||||
"model_name": "gpt-3.5-turbo", # model group name
|
||||
"model_name": "gpt-4o", # model group name
|
||||
"litellm_params": { # params for litellm completion/embedding call
|
||||
"model": "azure/chatgpt-v-2",
|
||||
"api_key": os.getenv("AZURE_API_KEY"),
|
||||
|
|
@ -1435,9 +1435,9 @@ model_list = [
|
|||
}
|
||||
},
|
||||
{
|
||||
"model_name": "gpt-3.5-turbo", # model group name
|
||||
"model_name": "gpt-4o", # model group name
|
||||
"litellm_params": { # params for litellm completion/embedding call
|
||||
"model": "gpt-3.5-turbo-1106",
|
||||
"model": "gpt-4o-1106",
|
||||
"api_key": os.getenv("OPENAI_API_KEY"),
|
||||
},
|
||||
},
|
||||
|
|
@ -1448,7 +1448,7 @@ router = Router(model_list=model_list, enable_pre_call_checks=True)
|
|||
text = "What is the meaning of 42?" * 5000
|
||||
|
||||
response = router.completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[
|
||||
{"role": "system", "content": text},
|
||||
{"role": "user", "content": "Who was Alexander?"},
|
||||
|
|
@ -1462,7 +1462,7 @@ print(f"response: {response}")
|
|||
|
||||
```python
|
||||
"""
|
||||
- Give 2 gpt-3.5-turbo deployments, in eu + non-eu regions
|
||||
- Give 2 gpt-4o deployments, in eu + non-eu regions
|
||||
- Make a call
|
||||
- Assert it picks the eu-region model
|
||||
"""
|
||||
|
|
@ -1472,7 +1472,7 @@ import os
|
|||
|
||||
model_list = [
|
||||
{
|
||||
"model_name": "gpt-3.5-turbo", # model group name
|
||||
"model_name": "gpt-4o", # model group name
|
||||
"litellm_params": { # params for litellm completion/embedding call
|
||||
"model": "azure/chatgpt-v-2",
|
||||
"api_key": os.getenv("AZURE_API_KEY"),
|
||||
|
|
@ -1485,9 +1485,9 @@ model_list = [
|
|||
}
|
||||
},
|
||||
{
|
||||
"model_name": "gpt-3.5-turbo", # model group name
|
||||
"model_name": "gpt-4o", # model group name
|
||||
"litellm_params": { # params for litellm completion/embedding call
|
||||
"model": "gpt-3.5-turbo-1106",
|
||||
"model": "gpt-4o-1106",
|
||||
"api_key": os.getenv("OPENAI_API_KEY"),
|
||||
},
|
||||
"model_info": {
|
||||
|
|
@ -1499,7 +1499,7 @@ model_list = [
|
|||
router = Router(model_list=model_list, enable_pre_call_checks=True)
|
||||
|
||||
response = router.completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Who was Alexander?"}],
|
||||
)
|
||||
|
||||
|
|
@ -1539,14 +1539,14 @@ async def test_acompletion_caching_on_router_caching_groups():
|
|||
litellm.set_verbose = True
|
||||
model_list = [
|
||||
{
|
||||
"model_name": "openai-gpt-3.5-turbo",
|
||||
"model_name": "openai-gpt-4o",
|
||||
"litellm_params": {
|
||||
"model": "gpt-3.5-turbo-0613",
|
||||
"model": "gpt-4o-0613",
|
||||
"api_key": os.getenv("OPENAI_API_KEY"),
|
||||
},
|
||||
},
|
||||
{
|
||||
"model_name": "azure-gpt-3.5-turbo",
|
||||
"model_name": "azure-gpt-4o",
|
||||
"litellm_params": {
|
||||
"model": "azure/chatgpt-v-2",
|
||||
"api_key": os.getenv("AZURE_API_KEY"),
|
||||
|
|
@ -1562,11 +1562,11 @@ async def test_acompletion_caching_on_router_caching_groups():
|
|||
start_time = time.time()
|
||||
router = Router(model_list=model_list,
|
||||
cache_responses=True,
|
||||
caching_groups=[("openai-gpt-3.5-turbo", "azure-gpt-3.5-turbo")])
|
||||
response1 = await router.acompletion(model="openai-gpt-3.5-turbo", messages=messages, temperature=1)
|
||||
caching_groups=[("openai-gpt-4o", "azure-gpt-4o")])
|
||||
response1 = await router.acompletion(model="openai-gpt-4o", messages=messages, temperature=1)
|
||||
print(f"response1: {response1}")
|
||||
await asyncio.sleep(1) # add cache is async, async sleep for cache to get set
|
||||
response2 = await router.acompletion(model="azure-gpt-3.5-turbo", messages=messages, temperature=1)
|
||||
response2 = await router.acompletion(model="azure-gpt-4o", messages=messages, temperature=1)
|
||||
assert response1.id == response2.id
|
||||
assert len(response1.choices[0].message.content) > 0
|
||||
assert response1.choices[0].message.content == response2.choices[0].message.content
|
||||
|
|
@ -1597,9 +1597,9 @@ import asyncio
|
|||
router = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "gpt-3.5-turbo",
|
||||
"model_name": "gpt-4o",
|
||||
"litellm_params": {
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"api_key": "bad_key",
|
||||
},
|
||||
}
|
||||
|
|
@ -1616,7 +1616,7 @@ async def main():
|
|||
|
||||
try:
|
||||
await router.acompletion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Hey, how's it going?"}],
|
||||
)
|
||||
except Exception as e:
|
||||
|
|
@ -1701,7 +1701,7 @@ You can also set default params for litellm completion/embedding calls. Here's h
|
|||
```python
|
||||
from litellm import Router
|
||||
|
||||
fallback_dict = {"gpt-3.5-turbo": "gpt-3.5-turbo-16k"}
|
||||
fallback_dict = {"gpt-4o": "gpt-4o-16k"}
|
||||
|
||||
router = Router(model_list=model_list,
|
||||
default_litellm_params={"context_window_fallback_dict": fallback_dict})
|
||||
|
|
@ -1710,7 +1710,7 @@ user_message = "Hello, whats the weather in San Francisco??"
|
|||
messages = [{"content": user_message, "role": "user"}]
|
||||
|
||||
# normal call
|
||||
response = router.completion(model="gpt-3.5-turbo", messages=messages)
|
||||
response = router.completion(model="gpt-4o", messages=messages)
|
||||
|
||||
print(f"response: {response}")
|
||||
```
|
||||
|
|
@ -1754,7 +1754,7 @@ router = Router(model_list=model_list, routing_strategy="simple-shuffle")
|
|||
|
||||
# router completion call
|
||||
response = router.completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{ "role": "user", "content": "Hi who are you"}]
|
||||
)
|
||||
```
|
||||
|
|
|
|||
|
|
@ -18,7 +18,7 @@ def my_custom_rule(input): # receives the model response
|
|||
|
||||
litellm.post_call_rules = [my_custom_rule] # have these be functions that can be called to fail a call
|
||||
|
||||
response = litellm.completion(model="gpt-3.5-turbo", messages=[{"role": "user",
|
||||
response = litellm.completion(model="gpt-4o", messages=[{"role": "user",
|
||||
"content": "Hey, how's it going?"}], fallbacks=["openrouter/gryphe/mythomax-l2-13b"])
|
||||
```
|
||||
|
||||
|
|
@ -63,7 +63,7 @@ def my_custom_rule(input): # receives the model response
|
|||
|
||||
litellm.pre_call_rules = [my_custom_rule] # have these be functions that can be called to fail a call
|
||||
|
||||
response = litellm.completion(model="gpt-3.5-turbo", messages=[{"role": "user", "content": "Hey, how's it going?"}])
|
||||
response = litellm.completion(model="gpt-4o", messages=[{"role": "user", "content": "Hey, how's it going?"}])
|
||||
```
|
||||
|
||||
### Example 2: Fallback to uncensored model if llm refuses to answer
|
||||
|
|
@ -84,6 +84,6 @@ def my_custom_rule(input): # receives the model response
|
|||
|
||||
litellm.post_call_rules = [my_custom_rule] # have these be functions that can be called to fail a call
|
||||
|
||||
response = litellm.completion(model="gpt-3.5-turbo", messages=[{"role": "user",
|
||||
response = litellm.completion(model="gpt-4o", messages=[{"role": "user",
|
||||
"content": "Hey, how's it going?"}], fallbacks=["openrouter/gryphe/mythomax-l2-13b"])
|
||||
```
|
||||
|
|
@ -32,9 +32,9 @@ from litellm import Router
|
|||
router = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "gpt-3.5-turbo",
|
||||
"model_name": "gpt-4o",
|
||||
"litellm_params": {
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"mock_response": "Hello world this is Macintosh!", # fakes the LLM API call
|
||||
"rpm": 1,
|
||||
},
|
||||
|
|
@ -47,7 +47,7 @@ router = Router(
|
|||
|
||||
try:
|
||||
_response = await router.acompletion( # 👈 ADDS TO QUEUE + POLLS + MAKES CALL
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Hey!"}],
|
||||
priority=0, # 👈 LOWER IS BETTER
|
||||
)
|
||||
|
|
@ -67,7 +67,7 @@ curl -X POST 'http://localhost:4000/chat/completions' \
|
|||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-D '{
|
||||
"model": "gpt-3.5-turbo-fake-model",
|
||||
"model": "gpt-4o-fake-model",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -89,7 +89,7 @@ client = openai.OpenAI(
|
|||
|
||||
# request sent to model set on litellm proxy, `litellm --model`
|
||||
response = client.chat.completions.create(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -118,9 +118,9 @@ from litellm import Router
|
|||
router = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "gpt-3.5-turbo",
|
||||
"model_name": "gpt-4o",
|
||||
"litellm_params": {
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"mock_response": "Hello world this is Macintosh!", # fakes the LLM API call
|
||||
"rpm": 1,
|
||||
},
|
||||
|
|
@ -134,7 +134,7 @@ router = Router(
|
|||
|
||||
try:
|
||||
_response = await router.acompletion( # 👈 ADDS TO QUEUE + POLLS + MAKES CALL
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Hey!"}],
|
||||
priority=0, # 👈 LOWER IS BETTER
|
||||
)
|
||||
|
|
@ -146,9 +146,9 @@ except Exception as e:
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo-fake-model
|
||||
- model_name: gpt-4o-fake-model
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
model: gpt-4o
|
||||
mock_response: "hello world!"
|
||||
api_key: my-good-key
|
||||
|
||||
|
|
@ -172,7 +172,7 @@ curl -X POST 'http://localhost:4000/queue/chat/completions' \
|
|||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-D '{
|
||||
"model": "gpt-3.5-turbo-fake-model",
|
||||
"model": "gpt-4o-fake-model",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
|
|||
|
|
@ -24,7 +24,7 @@ import TabItem from '@theme/TabItem';
|
|||
from litellm import text_completion
|
||||
|
||||
response = text_completion(
|
||||
model="gpt-3.5-turbo-instruct",
|
||||
model="gpt-4o-instruct",
|
||||
prompt="Say this is a test",
|
||||
max_tokens=7
|
||||
)
|
||||
|
|
@ -37,9 +37,9 @@ response = text_completion(
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo-instruct
|
||||
- model_name: gpt-4o-instruct
|
||||
litellm_params:
|
||||
model: text-completion-openai/gpt-3.5-turbo-instruct # The `text-completion-openai/` prefix will call openai.completions.create
|
||||
model: text-completion-openai/gpt-4o-instruct # The `text-completion-openai/` prefix will call openai.completions.create
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
- model_name: text-davinci-003
|
||||
litellm_params:
|
||||
|
|
@ -64,7 +64,7 @@ from openai import OpenAI
|
|||
client = OpenAI(api_key="<proxy-api-key>", base_url="http://0.0.0.0:4000")
|
||||
|
||||
response = client.completions.create(
|
||||
model="gpt-3.5-turbo-instruct",
|
||||
model="gpt-4o-instruct",
|
||||
prompt="Say this is a test",
|
||||
max_tokens=7
|
||||
)
|
||||
|
|
@ -80,7 +80,7 @@ curl --location 'http://0.0.0.0:4000/completions' \
|
|||
--header 'Content-Type: application/json' \
|
||||
--header 'Authorization: Bearer sk-1234' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo-instruct",
|
||||
"model": "gpt-4o-instruct",
|
||||
"prompt": "Say this is a test",
|
||||
"max_tokens": 7
|
||||
}'
|
||||
|
|
@ -133,7 +133,7 @@ Here's the exact JSON output format you can expect from completion calls:
|
|||
"id": "cmpl-uqkvlQyYK7bGYrRHQ0eXlWi7",
|
||||
"object": "text_completion",
|
||||
"created": 1589478378,
|
||||
"model": "gpt-3.5-turbo-instruct",
|
||||
"model": "gpt-4o-instruct",
|
||||
"system_fingerprint": "fp_44709d6fcb",
|
||||
"choices": [
|
||||
{
|
||||
|
|
@ -167,7 +167,7 @@ Here's the exact JSON output format you can expect from completion calls:
|
|||
"finish_reason": null
|
||||
}
|
||||
],
|
||||
"model": "gpt-3.5-turbo-instruct"
|
||||
"model": "gpt-4o-instruct"
|
||||
"system_fingerprint": "fp_44709d6fcb",
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -22,7 +22,7 @@ from litellm import Router
|
|||
|
||||
model_list = [
|
||||
{
|
||||
"model_name": "gpt-3.5-turbo",
|
||||
"model_name": "gpt-4o",
|
||||
"litellm_params": {
|
||||
"model": "azure/chatgpt-v-2",
|
||||
"api_key": "...",
|
||||
|
|
@ -40,9 +40,9 @@ model_list = [
|
|||
|
||||
router = Router(model_list=model_list)
|
||||
|
||||
# The request to "gpt-3.5-turbo" will trigger a background call to "gpt-4"
|
||||
# The request to "gpt-4o" will trigger a background call to "gpt-4"
|
||||
response = await router.acompletion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "How does traffic mirroring work?"}]
|
||||
)
|
||||
```
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue