docs: replace gpt-3.5-turbo with gpt-4o in completion guides

Co-authored-by: Krish Dholakia <krrishdholakia@gmail.com>
This commit is contained in:
Cursor Agent 2026-03-21 18:01:06 +00:00
parent 0cda3d4ff8
commit 6f10ddb843
No known key found for this signature in database
15 changed files with 56 additions and 56 deletions

View file

@ -216,8 +216,8 @@ Use `litellm.supports_audio_input(model="")` -> returns `True` if model can acce
assert litellm.supports_audio_output(model="gpt-4o-audio-preview") == True
assert litellm.supports_audio_input(model="gpt-4o-audio-preview") == True
assert litellm.supports_audio_output(model="gpt-3.5-turbo") == False
assert litellm.supports_audio_input(model="gpt-3.5-turbo") == False
assert litellm.supports_audio_output(model="gpt-4o") == False
assert litellm.supports_audio_input(model="gpt-4o") == False
```
</TabItem>

View file

@ -68,7 +68,7 @@ os.environ['OPENAI_API_KEY'] = ""
os.environ['COHERE_API_KEY'] = ""
response = batch_completion_models(
models=["gpt-3.5-turbo", "claude-instant-1.2", "command-nightly"],
models=["gpt-4o", "claude-instant-1.2", "command-nightly"],
messages=[{"role": "user", "content": "Hey, how's it going"}]
)
print(result)
@ -203,7 +203,7 @@ os.environ['OPENAI_API_KEY'] = ""
os.environ['COHERE_API_KEY'] = ""
responses = batch_completion_models_all_responses(
models=["gpt-3.5-turbo", "claude-instant-1.2", "command-nightly"],
models=["gpt-4o", "claude-instant-1.2", "command-nightly"],
messages=[{"role": "user", "content": "Hey, how's it going"}]
)
print(responses)
@ -259,7 +259,7 @@ print(responses)
"id": "chatcmpl-80szFnKHzCxObW0RqCMw1hWW1Icrq",
"object": "chat.completion",
"created": 1695222061,
"model": "gpt-3.5-turbo-0613",
"model": "gpt-4o-0613",
"choices": [
{
"index": 0,

View file

@ -210,7 +210,7 @@ client = openai.OpenAI(
)
response = client.chat.completions.create(
model="gpt-3.5-turbo",
model="gpt-4o",
messages = [
{
"role": "user",

View file

@ -5,7 +5,7 @@
Use `litellm.supports_function_calling(model="")` -> returns `True` if model supports Function calling, `False` if not
```python
assert litellm.supports_function_calling(model="gpt-3.5-turbo") == True
assert litellm.supports_function_calling(model="gpt-4o") == True
assert litellm.supports_function_calling(model="azure/gpt-4-1106-preview") == True
assert litellm.supports_function_calling(model="palm/chat-bison") == False
assert litellm.supports_function_calling(model="xai/grok-2-latest") == True
@ -24,7 +24,7 @@ assert litellm.supports_parallel_function_calling(model="gpt-4") == False
## Parallel Function calling
Parallel function calling is the model's ability to perform multiple function calls together, allowing the effects and results of these function calls to be resolved in parallel
## Quick Start - gpt-3.5-turbo-1106
## Quick Start - gpt-4o-1106
<a target="_blank" href="https://colab.research.google.com/github/BerriAI/litellm/blob/main/cookbook/Parallel_function_calling.ipynb">
<img src="https://colab.research.google.com/assets/colab-badge.svg" alt="Open In Colab"/>
</a>
@ -36,7 +36,7 @@ In this example we define a single function `get_current_weather`.
- Step 3: Send the model the output from running the `get_current_weather` function
### Full Code - Parallel function calling with `gpt-3.5-turbo-1106`
### Full Code - Parallel function calling with `gpt-4o-1106`
```python
import litellm
@ -84,7 +84,7 @@ def test_parallel_function_call():
}
]
response = litellm.completion(
model="gpt-3.5-turbo-1106",
model="gpt-4o-1106",
messages=messages,
tools=tools,
tool_choice="auto", # auto is default, but we'll be explicit
@ -122,7 +122,7 @@ def test_parallel_function_call():
}
) # extend conversation with function response
second_response = litellm.completion(
model="gpt-3.5-turbo-1106",
model="gpt-4o-1106",
messages=messages,
) # get a new response from the model where it can see the function response
print("\nSecond LLM response:\n", second_response)
@ -134,7 +134,7 @@ test_parallel_function_call()
```
### Explanation - Parallel function calling
Below is an explanation of what is happening in the code snippet above for Parallel function calling with `gpt-3.5-turbo-1106`
Below is an explanation of what is happening in the code snippet above for Parallel function calling with `gpt-4o-1106`
### Step1: litellm.completion() with `tools` set to `get_current_weather`
```python
import litellm
@ -178,7 +178,7 @@ tools = [
]
response = litellm.completion(
model="gpt-3.5-turbo-1106",
model="gpt-4o-1106",
messages=messages,
tools=tools,
tool_choice="auto", # auto is default, but we'll be explicit
@ -206,7 +206,7 @@ ModelResponse(
]))
],
created=1700319953,
model='gpt-3.5-turbo-1106',
model='gpt-4o-1106',
object='chat.completion',
system_fingerprint='fp_eeff13170a',
usage={'completion_tokens': 77, 'prompt_tokens': 88, 'total_tokens': 165},
@ -255,7 +255,7 @@ if tool_calls:
Once the functions are executed, send the model the information for each function call and its response. This allows the model to generate a new response considering the effects of the function calls.
```python
second_response = litellm.completion(
model="gpt-3.5-turbo-1106",
model="gpt-4o-1106",
messages=messages,
)
print("Second Response\n", second_response)
@ -270,7 +270,7 @@ ModelResponse(
message=Message(content="The current weather in San Francisco is 72°F, in Tokyo it's 10°C, and in Paris it's 22°C.", role='assistant'))
],
created=1700319955,
model='gpt-3.5-turbo-1106',
model='gpt-4o-1106',
object='chat.completion',
system_fingerprint='fp_eeff13170a',
usage={'completion_tokens': 28, 'prompt_tokens': 169, 'total_tokens': 197},
@ -415,7 +415,7 @@ functions = [
}
]
response = completion(model="gpt-3.5-turbo-0613", messages=messages, functions=functions)
response = completion(model="gpt-4o-0613", messages=messages, functions=functions)
print(response)
```
@ -499,7 +499,7 @@ def get_current_weather(location: str, unit: str):
functions = [litellm.utils.function_to_dict(get_current_weather)]
response = completion(model="gpt-3.5-turbo-0613", messages=messages, functions=functions)
response = completion(model="gpt-4o-0613", messages=messages, functions=functions)
print(response)
```

View file

@ -18,7 +18,7 @@ import litellm
# Works exactly as before
response = await litellm.acompletion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Hello!"}]
)
```
@ -39,7 +39,7 @@ session = aiohttp.ClientSession(
litellm.base_llm_aiohttp_handler = BaseLLMAIOHTTPHandler(client_session=session)
# All completions now use your session
response = await litellm.acompletion(model="gpt-3.5-turbo", messages=[...])
response = await litellm.acompletion(model="gpt-4o", messages=[...])
```
## Common Patterns
@ -69,7 +69,7 @@ app = FastAPI(lifespan=lifespan)
@app.post("/chat")
async def chat(messages: list[dict]):
return await litellm.acompletion(model="gpt-3.5-turbo", messages=messages)
return await litellm.acompletion(model="gpt-4o", messages=messages)
```
### Corporate Proxy

View file

@ -15,7 +15,7 @@ os.environ["OPENAI_API_KEY"] = "your-openai-key"
## SET MAX TOKENS - via completion()
response = litellm.completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{ "content": "Hello, how are you?","role": "user"}],
max_tokens=10
)

View file

@ -8,7 +8,7 @@ This will return a response object with a default response (works for streaming
```python
from litellm import completion
model = "gpt-3.5-turbo"
model = "gpt-4o"
messages = [{"role":"user", "content":"This is a test request"}]
completion(model=model, messages=messages, mock_response="It's simple to use and easy to get started")
@ -18,7 +18,7 @@ completion(model=model, messages=messages, mock_response="It's simple to use and
```python
from litellm import completion
model = "gpt-3.5-turbo"
model = "gpt-4o"
messages = [{"role": "user", "content": "Hey, I'm a mock request"}]
response = completion(model=model, messages=messages, stream=True, mock_response="It's simple to use and easy to get started")
for chunk in response:
@ -60,7 +60,7 @@ import pytest
def test_completion_openai():
try:
response = completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role":"user", "content":"Why is LiteLLM amazing?"}],
mock_response="LiteLLM is awesome"
)

View file

@ -1,6 +1,6 @@
# Model Alias
The model name you show an end-user might be different from the one you pass to LiteLLM - e.g. Displaying `GPT-3.5` while calling `gpt-3.5-turbo-16k` on the backend.
The model name you show an end-user might be different from the one you pass to LiteLLM - e.g. Displaying `GPT-3.5` while calling `gpt-4o-16k` on the backend.
LiteLLM simplifies this by letting you pass in a model alias mapping.
@ -18,7 +18,7 @@ litellm.model_alias_map = {
### Relevant Code
```python
model_alias_map = {
"GPT-3.5": "gpt-3.5-turbo-16k",
"GPT-3.5": "gpt-4o-16k",
"llama2": "replicate/llama-2-70b-chat:2796ee9483c3fd7aa2e171d38f4ca12251a30609463dcfd4cd76703f22e96cdf"
}
@ -37,7 +37,7 @@ os.environ["REPLICATE_API_KEY"] = "cohere key"
## set model alias map
model_alias_map = {
"GPT-3.5": "gpt-3.5-turbo-16k",
"GPT-3.5": "gpt-4o-16k",
"llama2": "replicate/llama-2-70b-chat:2796ee9483c3fd7aa2e171d38f4ca12251a30609463dcfd4cd76703f22e96cdf"
}
@ -45,7 +45,7 @@ litellm.model_alias_map = model_alias_map
messages = [{ "content": "Hello, how are you?","role": "user"}]
# call "gpt-3.5-turbo-16k"
# call "gpt-4o-16k"
response = completion(model="GPT-3.5", messages=messages)
# call replicate/llama-2-70b-chat:2796ee9483c3fd7aa2e171d38f4ca1...

View file

@ -700,7 +700,7 @@ response = client.chat.completions.with_raw_response.create(
"role": "user",
"content": "Say this is a test",
}],
model="gpt-3.5-turbo",
model="gpt-4o",
)
print(response.headers.get('x-litellm-response-cost'))

View file

@ -22,7 +22,7 @@ os.environ["OPENAI_API_KEY"] = "your-openai-key"
## SET MAX TOKENS - via completion()
response_1 = litellm.completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{ "content": "Hello, how are you?","role": "user"}],
max_tokens=10
)
@ -33,7 +33,7 @@ response_1_text = response_1.choices[0].message.content
litellm.OpenAIConfig(max_tokens=10)
response_2 = litellm.completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{ "content": "Hello, how are you?","role": "user"}],
)

View file

@ -25,7 +25,7 @@ messages = [{"content": user_message, "role": "user"}]
# normal call
response = completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=messages,
num_retries=2
)
@ -43,10 +43,10 @@ response = completion(
```python
from litellm import completion
fallback_dict = {"gpt-3.5-turbo": "gpt-3.5-turbo-16k"}
fallback_dict = {"gpt-4o": "gpt-4o-16k"}
messages = [{"content": "how does a court case get to the Supreme Court?" * 500, "role": "user"}]
completion(model="gpt-3.5-turbo", messages=messages, context_window_fallback_dict=fallback_dict)
completion(model="gpt-4o", messages=messages, context_window_fallback_dict=fallback_dict)
```
### Fallbacks - Switch Models/API Keys/API Bases (SDK)
@ -61,7 +61,7 @@ The `fallbacks` list should include the primary model you want to use, followed
#### switch models
```python
response = completion(model="bad-model", messages=messages,
fallbacks=["gpt-3.5-turbo" "command-nightly"])
fallbacks=["gpt-4o" "command-nightly"])
```
#### switch api keys/bases (E.g. azure deployment)
@ -84,12 +84,12 @@ Completion with 'bad-model': got exception Unable to map your input to a model.
completion call gpt-3.5-turbo
completion call gpt-4o
{
"id": "chatcmpl-7qTmVRuO3m3gIBg4aTmAumV1TmQhB",
"object": "chat.completion",
"created": 1692741891,
"model": "gpt-3.5-turbo-0613",
"model": "gpt-4o-0613",
"choices": [
{
"index": 0,

View file

@ -15,7 +15,7 @@ LiteLLM supports streaming the model response back by passing `stream=True` as a
```python
from litellm import completion
messages = [{"role": "user", "content": "Hey, how's it going?"}]
response = completion(model="gpt-3.5-turbo", messages=messages, stream=True)
response = completion(model="gpt-4o", messages=messages, stream=True)
for part in response:
print(part.choices[0].delta.content or "")
```
@ -27,7 +27,7 @@ LiteLLM also exposes a helper function to rebuild the complete streaming respons
```python
from litellm import completion
messages = [{"role": "user", "content": "Hey, how's it going?"}]
response = completion(model="gpt-3.5-turbo", messages=messages, stream=True)
response = completion(model="gpt-4o", messages=messages, stream=True)
for chunk in response:
chunks.append(chunk)
@ -45,7 +45,7 @@ import asyncio
async def test_get_response():
user_message = "Hello, how are you?"
messages = [{"content": user_message, "role": "user"}]
response = await acompletion(model="gpt-3.5-turbo", messages=messages)
response = await acompletion(model="gpt-4o", messages=messages)
return response
response = asyncio.run(test_get_response())
@ -66,7 +66,7 @@ async def completion_call():
try:
print("test acompletion + streaming")
response = await acompletion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"content": "Hello, how are you?", "role": "user"}],
stream=True
)
@ -106,7 +106,7 @@ chunks = [
"id": "chatcmpl-123",
"object": "chat.completion.chunk",
"created": 1694268190,
"model": "gpt-3.5-turbo-0125",
"model": "gpt-4o-0125",
"system_fingerprint": "fp_44709d6fcb",
"choices": [
{"index": 0, "delta": {"content": "How are you?"}, "finish_reason": "stop"}
@ -117,10 +117,10 @@ completion_stream = litellm.ModelResponseListIterator(model_responses=chunks)
response = litellm.CustomStreamWrapper(
completion_stream=completion_stream,
model="gpt-3.5-turbo",
model="gpt-4o",
custom_llm_provider="cached_response",
logging_obj=litellm.Logging(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Hey"}],
stream=True,
call_type="completion",

View file

@ -7,7 +7,7 @@ LiteLLM returns `response_cost` in all calls.
from litellm import completion
response = litellm.completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Hey, how's it going?"}],
mock_response="Hello world",
)
@ -49,7 +49,7 @@ from litellm import encode, decode
sample_text = "Hellö World, this is my input string!"
# openai encoding + decoding
openai_tokens = encode(model="gpt-3.5-turbo", text=sample_text)
openai_tokens = encode(model="gpt-4o", text=sample_text)
print(openai_tokens)
```
@ -62,8 +62,8 @@ from litellm import encode, decode
sample_text = "Hellö World, this is my input string!"
# openai encoding + decoding
openai_tokens = encode(model="gpt-3.5-turbo", text=sample_text)
openai_text = decode(model="gpt-3.5-turbo", tokens=openai_tokens)
openai_tokens = encode(model="gpt-4o", text=sample_text)
openai_text = decode(model="gpt-4o", tokens=openai_tokens)
print(openai_text)
```
@ -73,7 +73,7 @@ print(openai_text)
from litellm import token_counter
messages = [{"user": "role", "content": "Hey, how's it going"}]
print(token_counter(model="gpt-3.5-turbo", messages=messages))
print(token_counter(model="gpt-4o", messages=messages))
```
### 4. `create_pretrained_tokenizer` and `create_tokenizer`
@ -100,7 +100,7 @@ from litellm import cost_per_token
prompt_tokens = 5
completion_tokens = 10
prompt_tokens_cost_usd_dollar, completion_tokens_cost_usd_dollar = cost_per_token(model="gpt-3.5-turbo", prompt_tokens=prompt_tokens, completion_tokens=completion_tokens)
prompt_tokens_cost_usd_dollar, completion_tokens_cost_usd_dollar = cost_per_token(model="gpt-4o", prompt_tokens=prompt_tokens, completion_tokens=completion_tokens)
print(prompt_tokens_cost_usd_dollar, completion_tokens_cost_usd_dollar)
```
@ -134,13 +134,13 @@ print(formatted_string)
```
### 7. `get_max_tokens`
Input: Accepts a model name - e.g., gpt-3.5-turbo (to get a complete list, call litellm.model_list).
Input: Accepts a model name - e.g., gpt-4o (to get a complete list, call litellm.model_list).
Output: Returns the maximum number of tokens allowed for the given model
```python
from litellm import get_max_tokens
model = "gpt-3.5-turbo"
model = "gpt-4o"
print(get_max_tokens(model)) # Output: 4097
```
@ -152,7 +152,7 @@ print(get_max_tokens(model)) # Output: 4097
```python
from litellm import model_cost
print(model_cost) # {'gpt-3.5-turbo': {'max_tokens': 4000, 'input_cost_per_token': 1.5e-06, 'output_cost_per_token': 2e-06}, ...}
print(model_cost) # {'gpt-4o': {'max_tokens': 4000, 'input_cost_per_token': 1.5e-06, 'output_cost_per_token': 2e-06}, ...}
```
### 9. `register_model`

View file

@ -20,7 +20,7 @@ import os
os.environ["OPENAI_API_KEY"] = "your-api-key"
response = completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{ "content": "Hello, how are you?","role": "user"}]
)

View file

@ -120,7 +120,7 @@ Use `litellm.supports_vision(model="")` -> returns `True` if model supports `vis
```python
assert litellm.supports_vision(model="openai/gpt-4-vision-preview") == True
assert litellm.supports_vision(model="vertex_ai/gemini-1.0-pro-vision") == True
assert litellm.supports_vision(model="openai/gpt-3.5-turbo") == False
assert litellm.supports_vision(model="openai/gpt-4o") == False
assert litellm.supports_vision(model="xai/grok-2-vision-latest") == True
assert litellm.supports_vision(model="xai/grok-2-latest") == False
```