docs: replace gpt-3.5-turbo with gpt-4o in example docs

Co-authored-by: Krish Dholakia <krrishdholakia@gmail.com>
This commit is contained in:
Cursor Agent 2026-03-21 18:03:26 +00:00
parent 8e12f0a277
commit 2a2386071d
No known key found for this signature in database
13 changed files with 119 additions and 119 deletions

View file

@ -51,7 +51,7 @@ if not budget_manager.is_valid_user(user):
# check if a given call can be made
if budget_manager.get_current_cost(user=user) <= budget_manager.get_total_budget(user):
response = completion(model="gpt-3.5-turbo", messages=[{"role": "user", "content": "Hey, how's it going?"}])
response = completion(model="gpt-4o", messages=[{"role": "user", "content": "Hey, how's it going?"}])
budget_manager.update_cost(completion_obj=response, user=user)
else:
response = "Sorry - no budget!"
@ -72,7 +72,7 @@ budget_manager.create_budget(total_budget=10, user=user, duration="daily")
input_text = "hello world"
output_text = "it's a sunny day in san francisco"
model = "gpt-3.5-turbo"
model = "gpt-4o"
budget_manager.update_cost(user=user, model=model, input_text=input_text, output_text=output_text) # 👈
print(budget_manager.get_current_cost(user))
@ -108,7 +108,7 @@ if not budget_manager.is_valid_user(user):
# check if a given call can be made
if budget_manager.get_current_cost(user=user) <= budget_manager.get_total_budget(user):
response = completion(model="gpt-3.5-turbo", messages=[{"role": "user", "content": "Hey, how's it going?"}])
response = completion(model="gpt-4o", messages=[{"role": "user", "content": "Hey, how's it going?"}])
budget_manager.update_cost(completion_obj=response, user=user)
else:
response = "Sorry - no budget!"
@ -138,7 +138,7 @@ if not budget_manager.is_valid_user(user):
# check if a given call can be made
if budget_manager.get_current_cost(user=user) <= budget_manager.get_total_budget(user):
response = completion(model="gpt-3.5-turbo", messages=[{"role": "user", "content": "Hey, how's it going?"}])
response = completion(model="gpt-4o", messages=[{"role": "user", "content": "Hey, how's it going?"}])
budget_manager.update_cost(completion_obj=response, user=user)
else:
response = "Sorry - no budget!"

View file

@ -69,7 +69,7 @@ except openai.APITimeoutError as e:
import litellm
try:
response = litellm.completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[
{
"role": "user",

View file

@ -36,7 +36,7 @@ model_list = [
{
"model_name": "fake-openai-endpoint",
"litellm_params": {
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"api_key": "my-fake-key",
"api_base": "http://0.0.0.0:8080",
"rpm": 100
@ -45,7 +45,7 @@ model_list = [
{
"model_name": "fake-openai-endpoint",
"litellm_params": {
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"api_key": "my-fake-key",
"api_base": "http://0.0.0.0:8081",
"rpm": 100

View file

@ -42,7 +42,7 @@ curl http://localhost:4000/chat/completions \
-H "Content-Type: application/json" \
-H "Authorization: Bearer sk-1234" \
-d '{
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"messages": [
{"role": "user", "content": "My credit card is 4111-1111-1111-1111 and my email is john@example.com"}
],
@ -68,7 +68,7 @@ client = openai.OpenAI(
# This request will trigger MCP guardrails
response = client.chat.completions.create(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[
{"role": "user", "content": "Send an email to 555-123-4567 with my SSN 123-45-6789"}
],

View file

@ -18,12 +18,12 @@ When we have breaking changes (i.e. going from 1.x.x to 2.x.x), we will document
- *NEW* default exception - `APIConnectionError` (prev. `APIError`)
- litellm.get_max_tokens() now returns an int not a dict
```python
max_tokens = litellm.get_max_tokens("gpt-3.5-turbo") # returns an int not a dict
max_tokens = litellm.get_max_tokens("gpt-4o") # returns an int not a dict
assert max_tokens==4097
```
- Streaming - OpenAI Chunks now return `None` for empty stream chunks. This is how to process stream chunks with content
```python
response = litellm.completion(model="gpt-3.5-turbo", messages=messages, stream=True)
response = litellm.completion(model="gpt-4o", messages=messages, stream=True)
for part in response:
print(part.choices[0].delta.content or "")
```

View file

@ -11,9 +11,9 @@ Setup Prompt Injection Detection, Secret Detection on LiteLLM Proxy
```yaml
model_list:
- model_name: gpt-3.5-turbo
- model_name: gpt-4o
litellm_params:
model: openai/gpt-3.5-turbo
model: openai/gpt-4o
api_key: sk-xxxxxxx
litellm_settings:
@ -56,7 +56,7 @@ curl --location 'http://localhost:4000/chat/completions' \
--header 'Authorization: Bearer sk-1234' \
--header 'Content-Type: application/json' \
--data '{
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"messages": [
{
"role": "user",
@ -280,7 +280,7 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \
--header 'Content-Type: application/json' \
--header 'Authorization: Bearer $LITELLM_VIRTUAL_KEY' \
--data '{
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"messages": [
{
"role": "user",

View file

@ -15,7 +15,7 @@ os.environ["COHERE_API_KEY"] = "your-api-key"
messages = [{ "content": "Hello, how are you?","role": "user"}]
# openai call
response = completion(model="gpt-3.5-turbo", messages=messages)
response = completion(model="gpt-4o", messages=messages)
# cohere call
response = completion("command-nightly", messages)
@ -31,8 +31,8 @@ For a complete list of models/providers that you can call with LiteLLM, [check o
* OpenAI models - [OpenAI docs](./providers/openai.md)
* gpt-4
* gpt-3.5-turbo
* gpt-3.5-turbo-16k
* gpt-4o
* gpt-4o-16k
* Llama2 models - [TogetherAI docs](./providers/togetherai.md)
* togethercomputer/llama-2-70b-chat
* togethercomputer/llama-2-70b

View file

@ -520,7 +520,7 @@ assert litellm.supports_reasoning(model="anthropic/claude-3-7-sonnet-20250219")
assert litellm.supports_reasoning(model="deepseek/deepseek-chat") == True
# Example models that do not support reasoning
assert litellm.supports_reasoning(model="openai/gpt-3.5-turbo") == False
assert litellm.supports_reasoning(model="openai/gpt-4o") == False
```
</TabItem>

View file

@ -34,7 +34,7 @@ Loadbalance across multiple [azure](./providers/azure)/[bedrock](./providers/bed
from litellm import Router
model_list = [{ # list of model deployments
"model_name": "gpt-3.5-turbo", # model alias -> loadbalance between models with same `model_name`
"model_name": "gpt-4o", # model alias -> loadbalance between models with same `model_name`
"litellm_params": { # params for litellm completion/embedding call
"model": "azure/chatgpt-v-2", # actual model name
"api_key": os.getenv("AZURE_API_KEY"),
@ -42,7 +42,7 @@ model_list = [{ # list of model deployments
"api_base": os.getenv("AZURE_API_BASE")
}
}, {
"model_name": "gpt-3.5-turbo",
"model_name": "gpt-4o",
"litellm_params": { # params for litellm completion/embedding call
"model": "azure/chatgpt-functioncalling",
"api_key": os.getenv("AZURE_API_KEY"),
@ -50,9 +50,9 @@ model_list = [{ # list of model deployments
"api_base": os.getenv("AZURE_API_BASE")
}
}, {
"model_name": "gpt-3.5-turbo",
"model_name": "gpt-4o",
"litellm_params": { # params for litellm completion/embedding call
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"api_key": os.getenv("OPENAI_API_KEY"),
}
}, {
@ -76,8 +76,8 @@ model_list = [{ # list of model deployments
router = Router(model_list=model_list)
# openai.ChatCompletion.create replacement
# requests with model="gpt-3.5-turbo" will pick a deployment where model_name="gpt-3.5-turbo"
response = await router.acompletion(model="gpt-3.5-turbo",
# requests with model="gpt-4o" will pick a deployment where model_name="gpt-4o"
response = await router.acompletion(model="gpt-4o",
messages=[{"role": "user", "content": "Hey, how's it going?"}])
print(response)
@ -101,17 +101,17 @@ See detailed proxy loadbalancing/fallback docs [here](./proxy/reliability.md)
1. Setup model_list with multiple deployments
```yaml
model_list:
- model_name: gpt-3.5-turbo
- model_name: gpt-4o
litellm_params:
model: azure/<your-deployment-name>
api_base: <your-azure-endpoint>
api_key: <your-azure-api-key>
- model_name: gpt-3.5-turbo
- model_name: gpt-4o
litellm_params:
model: azure/gpt-turbo-small-ca
api_base: https://my-endpoint-canada-berri992.openai.azure.com/
api_key: <your-azure-api-key>
- model_name: gpt-3.5-turbo
- model_name: gpt-4o
litellm_params:
model: azure/gpt-turbo-large
api_base: https://openai-france-1234.openai.azure.com/
@ -131,7 +131,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
-H 'Content-Type: application/json' \
-H 'Authorization: Bearer sk-1234' \
-d '{
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"messages": [
{"role": "user", "content": "Hi there!"}
],
@ -174,14 +174,14 @@ You can also set a `weight` param, to specify which model should get picked when
```yaml
model_list:
- model_name: gpt-3.5-turbo
- model_name: gpt-4o
litellm_params:
model: azure/chatgpt-v-2
api_key: os.environ/AZURE_API_KEY
api_version: os.environ/AZURE_API_VERSION
api_base: os.environ/AZURE_API_BASE
rpm: 900
- model_name: gpt-3.5-turbo
- model_name: gpt-4o
litellm_params:
model: azure/chatgpt-functioncalling
api_key: os.environ/AZURE_API_KEY
@ -197,7 +197,7 @@ from litellm import Router
import asyncio
model_list = [{ # list of model deployments
"model_name": "gpt-3.5-turbo", # model alias
"model_name": "gpt-4o", # model alias
"litellm_params": { # params for litellm completion/embedding call
"model": "azure/chatgpt-v-2", # actual model name
"api_key": os.getenv("AZURE_API_KEY"),
@ -206,7 +206,7 @@ model_list = [{ # list of model deployments
"rpm": 900, # requests per minute for this API
}
}, {
"model_name": "gpt-3.5-turbo",
"model_name": "gpt-4o",
"litellm_params": { # params for litellm completion/embedding call
"model": "azure/chatgpt-functioncalling",
"api_key": os.getenv("AZURE_API_KEY"),
@ -220,7 +220,7 @@ model_list = [{ # list of model deployments
router = Router(model_list=model_list, routing_strategy="simple-shuffle")
async def router_acompletion():
response = await router.acompletion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Hey, how's it going?"}]
)
print(response)
@ -236,14 +236,14 @@ asyncio.run(router_acompletion())
```yaml
model_list:
- model_name: gpt-3.5-turbo
- model_name: gpt-4o
litellm_params:
model: azure/chatgpt-v-2
api_key: os.environ/AZURE_API_KEY
api_version: os.environ/AZURE_API_VERSION
api_base: os.environ/AZURE_API_BASE
weight: 9
- model_name: gpt-3.5-turbo
- model_name: gpt-4o
litellm_params:
model: azure/chatgpt-functioncalling
api_key: os.environ/AZURE_API_KEY
@ -259,7 +259,7 @@ from litellm import Router
import asyncio
model_list = [{
"model_name": "gpt-3.5-turbo", # model alias
"model_name": "gpt-4o", # model alias
"litellm_params": {
"model": "azure/chatgpt-v-2", # actual model name
"api_key": os.getenv("AZURE_API_KEY"),
@ -268,7 +268,7 @@ model_list = [{
"weight": 9, # pick this 90% of the time
}
}, {
"model_name": "gpt-3.5-turbo",
"model_name": "gpt-4o",
"litellm_params": {
"model": "azure/chatgpt-functioncalling",
"api_key": os.getenv("AZURE_API_KEY"),
@ -282,7 +282,7 @@ model_list = [{
router = Router(model_list=model_list, routing_strategy="simple-shuffle")
async def router_acompletion():
response = await router.acompletion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Hey, how's it going?"}]
)
print(response)
@ -319,7 +319,7 @@ from litellm import Router
model_list = [{ # list of model deployments
"model_name": "gpt-3.5-turbo", # model alias
"model_name": "gpt-4o", # model alias
"litellm_params": { # params for litellm completion/embedding call
"model": "azure/chatgpt-v-2", # actual model name
"api_key": os.getenv("AZURE_API_KEY"),
@ -329,7 +329,7 @@ model_list = [{ # list of model deployments
"rpm": 10000,
},
}, {
"model_name": "gpt-3.5-turbo",
"model_name": "gpt-4o",
"litellm_params": { # params for litellm completion/embedding call
"model": "azure/chatgpt-functioncalling",
"api_key": os.getenv("AZURE_API_KEY"),
@ -339,9 +339,9 @@ model_list = [{ # list of model deployments
"rpm": 1000,
},
}, {
"model_name": "gpt-3.5-turbo",
"model_name": "gpt-4o",
"litellm_params": { # params for litellm completion/embedding call
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"api_key": os.getenv("OPENAI_API_KEY"),
"tpm": 100000,
"rpm": 1000,
@ -355,7 +355,7 @@ router = Router(model_list=model_list,
enable_pre_call_checks=True, # enables router rate limits for concurrent calls
)
response = await router.acompletion(model="gpt-3.5-turbo",
response = await router.acompletion(model="gpt-4o",
messages=[{"role": "user", "content": "Hey, how's it going?"}]
print(response)
@ -367,7 +367,7 @@ print(response)
```yaml
model_list:
- model_name: gpt-3.5-turbo # model alias
- model_name: gpt-4o # model alias
litellm_params: # params for litellm completion/embedding call
model: azure/chatgpt-v-2 # actual model name
api_key: os.environ/AZURE_API_KEY
@ -375,9 +375,9 @@ model_list:
api_base: os.environ/AZURE_API_BASE
tpm: 100000
rpm: 10000
- model_name: gpt-3.5-turbo
- model_name: gpt-4o
litellm_params: # params for litellm completion/embedding call
model: gpt-3.5-turbo
model: gpt-4o
api_key: os.getenv(OPENAI_API_KEY)
tpm: 100000
rpm: 1000
@ -406,7 +406,7 @@ curl --location 'http://localhost:4000/v1/chat/completions' \
--header 'Content-Type: application/json' \
--header 'Authorization: Bearer sk-1234' \
--data '{
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"messages": [{"role": "user", "content": "Hey, how's it going?"}]
}'
```
@ -524,7 +524,7 @@ from litellm import Router
model_list = [{ # list of model deployments
"model_name": "gpt-3.5-turbo", # model alias
"model_name": "gpt-4o", # model alias
"litellm_params": { # params for litellm completion/embedding call
"model": "azure/chatgpt-v-2", # actual model name
"api_key": os.getenv("AZURE_API_KEY"),
@ -534,7 +534,7 @@ model_list = [{ # list of model deployments
"tpm": 100000,
"rpm": 10000,
}, {
"model_name": "gpt-3.5-turbo",
"model_name": "gpt-4o",
"litellm_params": { # params for litellm completion/embedding call
"model": "azure/chatgpt-functioncalling",
"api_key": os.getenv("AZURE_API_KEY"),
@ -544,9 +544,9 @@ model_list = [{ # list of model deployments
"tpm": 100000,
"rpm": 1000,
}, {
"model_name": "gpt-3.5-turbo",
"model_name": "gpt-4o",
"litellm_params": { # params for litellm completion/embedding call
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"api_key": os.getenv("OPENAI_API_KEY"),
},
"tpm": 100000,
@ -560,7 +560,7 @@ router = Router(model_list=model_list,
enable_pre_call_check=True, # enables router rate limits for concurrent calls
)
response = await router.acompletion(model="gpt-3.5-turbo",
response = await router.acompletion(model="gpt-4o",
messages=[{"role": "user", "content": "Hey, how's it going?"}]
print(response)
@ -580,7 +580,7 @@ from litellm import Router
import asyncio
model_list = [{ # list of model deployments
"model_name": "gpt-3.5-turbo", # model alias
"model_name": "gpt-4o", # model alias
"litellm_params": { # params for litellm completion/embedding call
"model": "azure/chatgpt-v-2", # actual model name
"api_key": os.getenv("AZURE_API_KEY"),
@ -588,7 +588,7 @@ model_list = [{ # list of model deployments
"api_base": os.getenv("AZURE_API_BASE"),
}
}, {
"model_name": "gpt-3.5-turbo",
"model_name": "gpt-4o",
"litellm_params": { # params for litellm completion/embedding call
"model": "azure/chatgpt-functioncalling",
"api_key": os.getenv("AZURE_API_KEY"),
@ -596,9 +596,9 @@ model_list = [{ # list of model deployments
"api_base": os.getenv("AZURE_API_BASE"),
}
}, {
"model_name": "gpt-3.5-turbo",
"model_name": "gpt-4o",
"litellm_params": { # params for litellm completion/embedding call
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"api_key": os.getenv("OPENAI_API_KEY"),
}
}]
@ -607,7 +607,7 @@ model_list = [{ # list of model deployments
router = Router(model_list=model_list, routing_strategy="least-busy")
async def router_acompletion():
response = await router.acompletion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Hey, how's it going?"}]
)
print(response)
@ -750,12 +750,12 @@ import asyncio
model_list = [
{
"model_name": "gpt-3.5-turbo",
"model_name": "gpt-4o",
"litellm_params": {"model": "gpt-4"},
"model_info": {"id": "openai-gpt-4"},
},
{
"model_name": "gpt-3.5-turbo",
"model_name": "gpt-4o",
"litellm_params": {"model": "groq/llama3-8b-8192"},
"model_info": {"id": "groq-llama"},
},
@ -765,7 +765,7 @@ model_list = [
router = Router(model_list=model_list, routing_strategy="cost-based-routing")
async def router_acompletion():
response = await router.acompletion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Hey, how's it going?"}]
)
print(response)
@ -785,7 +785,7 @@ Set `litellm_params["input_cost_per_token"]` and `litellm_params["output_cost_pe
```python
model_list = [
{
"model_name": "gpt-3.5-turbo",
"model_name": "gpt-4o",
"litellm_params": {
"model": "azure/chatgpt-v-2",
"input_cost_per_token": 0.00003,
@ -794,7 +794,7 @@ model_list = [
"model_info": {"id": "chatgpt-v-experimental"},
},
{
"model_name": "gpt-3.5-turbo",
"model_name": "gpt-4o",
"litellm_params": {
"model": "azure/chatgpt-v-1",
"input_cost_per_token": 0.000000001,
@ -803,7 +803,7 @@ model_list = [
"model_info": {"id": "chatgpt-v-1"},
},
{
"model_name": "gpt-3.5-turbo",
"model_name": "gpt-4o",
"litellm_params": {
"model": "azure/chatgpt-v-5",
"input_cost_per_token": 10,
@ -816,7 +816,7 @@ model_list = [
router = Router(model_list=model_list, routing_strategy="cost-based-routing")
async def router_acompletion():
response = await router.acompletion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Hey, how's it going?"}]
)
print(response)
@ -932,7 +932,7 @@ model_list = [
router = Router(model_list=model_list, routing_strategy="cost-based-routing")
response = await router.acompletion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Hey, how's it going?"}]
)
print(response)
@ -1007,7 +1007,7 @@ user_message = "Hello, whats the weather in San Francisco??"
messages = [{"content": user_message, "role": "user"}]
# normal call
response = router.completion(model="gpt-3.5-turbo", messages=messages)
response = router.completion(model="gpt-4o", messages=messages)
print(f"response: {response}")
```
@ -1190,7 +1190,7 @@ user_message = "Hello, whats the weather in San Francisco??"
messages = [{"content": user_message, "role": "user"}]
# normal call
response = router.completion(model="gpt-3.5-turbo", messages=messages)
response = router.completion(model="gpt-4o", messages=messages)
print(f"response: {response}")
```
@ -1209,7 +1209,7 @@ user_message = "Hello, whats the weather in San Francisco??"
messages = [{"content": user_message, "role": "user"}]
# normal call
response = router.completion(model="gpt-3.5-turbo", messages=messages)
response = router.completion(model="gpt-4o", messages=messages)
print(f"response: {response}")
```
@ -1260,7 +1260,7 @@ allowed_fails_policy = AllowedFailsPolicy(
router = litellm.Router(
model_list=[
{
"model_name": "gpt-3.5-turbo", # openai model name
"model_name": "gpt-4o", # openai model name
"litellm_params": { # params for litellm completion/embedding call
"model": "azure/chatgpt-v-2",
"api_key": os.getenv("AZURE_API_KEY"),
@ -1374,7 +1374,7 @@ For 'eu-region' filtering, Set 'region_name' of deployment.
```python
model_list = [
{
"model_name": "gpt-3.5-turbo", # model group name
"model_name": "gpt-4o", # model group name
"litellm_params": { # params for litellm completion/embedding call
"model": "azure/chatgpt-v-2",
"api_key": os.getenv("AZURE_API_KEY"),
@ -1385,9 +1385,9 @@ model_list = [
},
},
{
"model_name": "gpt-3.5-turbo", # model group name
"model_name": "gpt-4o", # model group name
"litellm_params": { # params for litellm completion/embedding call
"model": "gpt-3.5-turbo-1106",
"model": "gpt-4o-1106",
"api_key": os.getenv("OPENAI_API_KEY"),
},
},
@ -1413,7 +1413,7 @@ router = Router(model_list=model_list, enable_pre_call_checks=True)
```python
"""
- Give a gpt-3.5-turbo model group with different context windows (4k vs. 16k)
- Give a gpt-4o model group with different context windows (4k vs. 16k)
- Send a 5k prompt
- Assert it works
"""
@ -1422,7 +1422,7 @@ import os
model_list = [
{
"model_name": "gpt-3.5-turbo", # model group name
"model_name": "gpt-4o", # model group name
"litellm_params": { # params for litellm completion/embedding call
"model": "azure/chatgpt-v-2",
"api_key": os.getenv("AZURE_API_KEY"),
@ -1435,9 +1435,9 @@ model_list = [
}
},
{
"model_name": "gpt-3.5-turbo", # model group name
"model_name": "gpt-4o", # model group name
"litellm_params": { # params for litellm completion/embedding call
"model": "gpt-3.5-turbo-1106",
"model": "gpt-4o-1106",
"api_key": os.getenv("OPENAI_API_KEY"),
},
},
@ -1448,7 +1448,7 @@ router = Router(model_list=model_list, enable_pre_call_checks=True)
text = "What is the meaning of 42?" * 5000
response = router.completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[
{"role": "system", "content": text},
{"role": "user", "content": "Who was Alexander?"},
@ -1462,7 +1462,7 @@ print(f"response: {response}")
```python
"""
- Give 2 gpt-3.5-turbo deployments, in eu + non-eu regions
- Give 2 gpt-4o deployments, in eu + non-eu regions
- Make a call
- Assert it picks the eu-region model
"""
@ -1472,7 +1472,7 @@ import os
model_list = [
{
"model_name": "gpt-3.5-turbo", # model group name
"model_name": "gpt-4o", # model group name
"litellm_params": { # params for litellm completion/embedding call
"model": "azure/chatgpt-v-2",
"api_key": os.getenv("AZURE_API_KEY"),
@ -1485,9 +1485,9 @@ model_list = [
}
},
{
"model_name": "gpt-3.5-turbo", # model group name
"model_name": "gpt-4o", # model group name
"litellm_params": { # params for litellm completion/embedding call
"model": "gpt-3.5-turbo-1106",
"model": "gpt-4o-1106",
"api_key": os.getenv("OPENAI_API_KEY"),
},
"model_info": {
@ -1499,7 +1499,7 @@ model_list = [
router = Router(model_list=model_list, enable_pre_call_checks=True)
response = router.completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Who was Alexander?"}],
)
@ -1539,14 +1539,14 @@ async def test_acompletion_caching_on_router_caching_groups():
litellm.set_verbose = True
model_list = [
{
"model_name": "openai-gpt-3.5-turbo",
"model_name": "openai-gpt-4o",
"litellm_params": {
"model": "gpt-3.5-turbo-0613",
"model": "gpt-4o-0613",
"api_key": os.getenv("OPENAI_API_KEY"),
},
},
{
"model_name": "azure-gpt-3.5-turbo",
"model_name": "azure-gpt-4o",
"litellm_params": {
"model": "azure/chatgpt-v-2",
"api_key": os.getenv("AZURE_API_KEY"),
@ -1562,11 +1562,11 @@ async def test_acompletion_caching_on_router_caching_groups():
start_time = time.time()
router = Router(model_list=model_list,
cache_responses=True,
caching_groups=[("openai-gpt-3.5-turbo", "azure-gpt-3.5-turbo")])
response1 = await router.acompletion(model="openai-gpt-3.5-turbo", messages=messages, temperature=1)
caching_groups=[("openai-gpt-4o", "azure-gpt-4o")])
response1 = await router.acompletion(model="openai-gpt-4o", messages=messages, temperature=1)
print(f"response1: {response1}")
await asyncio.sleep(1) # add cache is async, async sleep for cache to get set
response2 = await router.acompletion(model="azure-gpt-3.5-turbo", messages=messages, temperature=1)
response2 = await router.acompletion(model="azure-gpt-4o", messages=messages, temperature=1)
assert response1.id == response2.id
assert len(response1.choices[0].message.content) > 0
assert response1.choices[0].message.content == response2.choices[0].message.content
@ -1597,9 +1597,9 @@ import asyncio
router = Router(
model_list=[
{
"model_name": "gpt-3.5-turbo",
"model_name": "gpt-4o",
"litellm_params": {
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"api_key": "bad_key",
},
}
@ -1616,7 +1616,7 @@ async def main():
try:
await router.acompletion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Hey, how's it going?"}],
)
except Exception as e:
@ -1701,7 +1701,7 @@ You can also set default params for litellm completion/embedding calls. Here's h
```python
from litellm import Router
fallback_dict = {"gpt-3.5-turbo": "gpt-3.5-turbo-16k"}
fallback_dict = {"gpt-4o": "gpt-4o-16k"}
router = Router(model_list=model_list,
default_litellm_params={"context_window_fallback_dict": fallback_dict})
@ -1710,7 +1710,7 @@ user_message = "Hello, whats the weather in San Francisco??"
messages = [{"content": user_message, "role": "user"}]
# normal call
response = router.completion(model="gpt-3.5-turbo", messages=messages)
response = router.completion(model="gpt-4o", messages=messages)
print(f"response: {response}")
```
@ -1754,7 +1754,7 @@ router = Router(model_list=model_list, routing_strategy="simple-shuffle")
# router completion call
response = router.completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{ "role": "user", "content": "Hi who are you"}]
)
```

View file

@ -18,7 +18,7 @@ def my_custom_rule(input): # receives the model response
litellm.post_call_rules = [my_custom_rule] # have these be functions that can be called to fail a call
response = litellm.completion(model="gpt-3.5-turbo", messages=[{"role": "user",
response = litellm.completion(model="gpt-4o", messages=[{"role": "user",
"content": "Hey, how's it going?"}], fallbacks=["openrouter/gryphe/mythomax-l2-13b"])
```
@ -63,7 +63,7 @@ def my_custom_rule(input): # receives the model response
litellm.pre_call_rules = [my_custom_rule] # have these be functions that can be called to fail a call
response = litellm.completion(model="gpt-3.5-turbo", messages=[{"role": "user", "content": "Hey, how's it going?"}])
response = litellm.completion(model="gpt-4o", messages=[{"role": "user", "content": "Hey, how's it going?"}])
```
### Example 2: Fallback to uncensored model if llm refuses to answer
@ -84,6 +84,6 @@ def my_custom_rule(input): # receives the model response
litellm.post_call_rules = [my_custom_rule] # have these be functions that can be called to fail a call
response = litellm.completion(model="gpt-3.5-turbo", messages=[{"role": "user",
response = litellm.completion(model="gpt-4o", messages=[{"role": "user",
"content": "Hey, how's it going?"}], fallbacks=["openrouter/gryphe/mythomax-l2-13b"])
```

View file

@ -32,9 +32,9 @@ from litellm import Router
router = Router(
model_list=[
{
"model_name": "gpt-3.5-turbo",
"model_name": "gpt-4o",
"litellm_params": {
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"mock_response": "Hello world this is Macintosh!", # fakes the LLM API call
"rpm": 1,
},
@ -47,7 +47,7 @@ router = Router(
try:
_response = await router.acompletion( # 👈 ADDS TO QUEUE + POLLS + MAKES CALL
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Hey!"}],
priority=0, # 👈 LOWER IS BETTER
)
@ -67,7 +67,7 @@ curl -X POST 'http://localhost:4000/chat/completions' \
-H 'Content-Type: application/json' \
-H 'Authorization: Bearer sk-1234' \
-D '{
"model": "gpt-3.5-turbo-fake-model",
"model": "gpt-4o-fake-model",
"messages": [
{
"role": "user",
@ -89,7 +89,7 @@ client = openai.OpenAI(
# request sent to model set on litellm proxy, `litellm --model`
response = client.chat.completions.create(
model="gpt-3.5-turbo",
model="gpt-4o",
messages = [
{
"role": "user",
@ -118,9 +118,9 @@ from litellm import Router
router = Router(
model_list=[
{
"model_name": "gpt-3.5-turbo",
"model_name": "gpt-4o",
"litellm_params": {
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"mock_response": "Hello world this is Macintosh!", # fakes the LLM API call
"rpm": 1,
},
@ -134,7 +134,7 @@ router = Router(
try:
_response = await router.acompletion( # 👈 ADDS TO QUEUE + POLLS + MAKES CALL
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Hey!"}],
priority=0, # 👈 LOWER IS BETTER
)
@ -146,9 +146,9 @@ except Exception as e:
```yaml
model_list:
- model_name: gpt-3.5-turbo-fake-model
- model_name: gpt-4o-fake-model
litellm_params:
model: gpt-3.5-turbo
model: gpt-4o
mock_response: "hello world!"
api_key: my-good-key
@ -172,7 +172,7 @@ curl -X POST 'http://localhost:4000/queue/chat/completions' \
-H 'Content-Type: application/json' \
-H 'Authorization: Bearer sk-1234' \
-D '{
"model": "gpt-3.5-turbo-fake-model",
"model": "gpt-4o-fake-model",
"messages": [
{
"role": "user",

View file

@ -24,7 +24,7 @@ import TabItem from '@theme/TabItem';
from litellm import text_completion
response = text_completion(
model="gpt-3.5-turbo-instruct",
model="gpt-4o-instruct",
prompt="Say this is a test",
max_tokens=7
)
@ -37,9 +37,9 @@ response = text_completion(
```yaml
model_list:
- model_name: gpt-3.5-turbo-instruct
- model_name: gpt-4o-instruct
litellm_params:
model: text-completion-openai/gpt-3.5-turbo-instruct # The `text-completion-openai/` prefix will call openai.completions.create
model: text-completion-openai/gpt-4o-instruct # The `text-completion-openai/` prefix will call openai.completions.create
api_key: os.environ/OPENAI_API_KEY
- model_name: text-davinci-003
litellm_params:
@ -64,7 +64,7 @@ from openai import OpenAI
client = OpenAI(api_key="<proxy-api-key>", base_url="http://0.0.0.0:4000")
response = client.completions.create(
model="gpt-3.5-turbo-instruct",
model="gpt-4o-instruct",
prompt="Say this is a test",
max_tokens=7
)
@ -80,7 +80,7 @@ curl --location 'http://0.0.0.0:4000/completions' \
--header 'Content-Type: application/json' \
--header 'Authorization: Bearer sk-1234' \
--data '{
"model": "gpt-3.5-turbo-instruct",
"model": "gpt-4o-instruct",
"prompt": "Say this is a test",
"max_tokens": 7
}'
@ -133,7 +133,7 @@ Here's the exact JSON output format you can expect from completion calls:
"id": "cmpl-uqkvlQyYK7bGYrRHQ0eXlWi7",
"object": "text_completion",
"created": 1589478378,
"model": "gpt-3.5-turbo-instruct",
"model": "gpt-4o-instruct",
"system_fingerprint": "fp_44709d6fcb",
"choices": [
{
@ -167,7 +167,7 @@ Here's the exact JSON output format you can expect from completion calls:
"finish_reason": null
}
],
"model": "gpt-3.5-turbo-instruct"
"model": "gpt-4o-instruct"
"system_fingerprint": "fp_44709d6fcb",
}

View file

@ -22,7 +22,7 @@ from litellm import Router
model_list = [
{
"model_name": "gpt-3.5-turbo",
"model_name": "gpt-4o",
"litellm_params": {
"model": "azure/chatgpt-v-2",
"api_key": "...",
@ -40,9 +40,9 @@ model_list = [
router = Router(model_list=model_list)
# The request to "gpt-3.5-turbo" will trigger a background call to "gpt-4"
# The request to "gpt-4o" will trigger a background call to "gpt-4"
response = await router.acompletion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "How does traffic mirroring work?"}]
)
```