docs: replace gpt-3.5-turbo with gpt-4o in provider and guide examples

Update model strings across OpenRouter, Azure, caching, LangChain, Letta,
and related docs using StrReplace (replace_all) for consistency with gpt-4o.

Co-authored-by: Krish Dholakia <krrishdholakia@gmail.com>
This commit is contained in:
Cursor Agent 2026-03-21 18:03:31 +00:00
parent 2a2386071d
commit c4c2afc579
No known key found for this signature in database
14 changed files with 93 additions and 93 deletions

View file

@ -321,7 +321,7 @@ curl -X POST 'http://localhost:4000/{my_endpoint}' \
-H 'Content-Type: application/json' \
-H 'Authorization: Bearer your-api-key' \
-d '{
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"messages": [{"role": "user", "content": "Hello"}],
"guardrails": ["test"]
}'

View file

@ -131,9 +131,9 @@ Add to `config.yaml`:
```yaml
model_list:
- model_name: gpt-3.5-turbo
- model_name: gpt-4o
litellm_params:
model: openai/gpt-3.5-turbo
model: openai/gpt-4o
api_key: os.environ/OPENAI_API_KEY
prompts:

View file

@ -39,11 +39,11 @@ litellm.cache = Cache(type="redis", host=<host>, port=<port>, password=<password
# Make completion calls
response1 = completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Tell me a joke."}]
)
response2 = completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Tell me a joke."}]
)
@ -77,11 +77,11 @@ litellm.cache = RedisClusterCache(
# Make completion calls
response1 = completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Tell me a joke."}]
)
response2 = completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Tell me a joke."}]
)
@ -132,11 +132,11 @@ from litellm.caching.caching import Cache
litellm.cache = Cache(type="gcs", gcs_bucket_name="my-cache-bucket", gcs_path_service_account="/path/to/service_account.json")
response1 = completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Tell me a joke."}]
)
response2 = completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Tell me a joke."}]
)
@ -170,11 +170,11 @@ litellm.cache = Cache(type="s3", s3_bucket_name="cache-bucket-litellm", s3_regio
# Make completion calls
response1 = completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Tell me a joke."}]
)
response2 = completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Tell me a joke."}]
)
@ -201,11 +201,11 @@ litellm.cache = Cache(type="azure-blob", azure_account_url="https://example.blob
# Make completion calls
response1 = completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Tell me a joke."}]
)
response2 = completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Tell me a joke."}]
)
@ -244,7 +244,7 @@ litellm.cache = Cache(
redis_semantic_cache_embedding_model="text-embedding-ada-002", # this model is passed to litellm.embedding(), any litellm.embedding() model is supported here
)
response1 = completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[
{
"role": "user",
@ -258,7 +258,7 @@ print(f"response1: {response1}")
random_number = random.randint(1, 100000)
response2 = completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[
{
"role": "user",
@ -301,7 +301,7 @@ litellm.cache = Cache(
)
response1 = completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[
{
"role": "user",
@ -315,7 +315,7 @@ print(f"response1: {response1}")
random_number = random.randint(1, 100000)
response2 = completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[
{
"role": "user",
@ -343,12 +343,12 @@ litellm.cache = Cache()
# Make completion calls
response1 = completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Tell me a joke."}],
caching=True
)
response2 = completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Tell me a joke."}],
caching=True
)
@ -379,12 +379,12 @@ litellm.cache = Cache(type="disk")
# Make completion calls
response1 = completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Tell me a joke."}],
caching=True
)
response2 = completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Tell me a joke."}],
caching=True
)
@ -416,7 +416,7 @@ Example usage `no-cache` - When `True`, Will not return a cached response
```python
response = litellm.completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[
{
"role": "user",
@ -435,7 +435,7 @@ Example usage `no-store` - When `True`, Will not cache the response.
```python
response = litellm.completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[
{
"role": "user",
@ -453,7 +453,7 @@ Example usage `ttl` - cache the response for 10 seconds
```python
response = litellm.completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[
{
"role": "user",
@ -471,7 +471,7 @@ Example usage `s-maxage` - Will only accept cached responses for 60 seconds
```python
response = litellm.completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[
{
"role": "user",

View file

@ -11,13 +11,13 @@ litellm.cache = Cache(type="hosted") # init cache to use api.litellm.ai
# Make completion calls
response1 = completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Tell me a joke."}]
caching=True
)
response2 = completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Tell me a joke."}],
caching=True
)
@ -59,7 +59,7 @@ litellm.cache = Cache(type="hosted")
# Make completion calls
response1 = completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Tell me a joke."}],
stream=True,
caching=True)
@ -69,7 +69,7 @@ for chunk in response1:
time.sleep(1) # cache is updated asynchronously
response2 = completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Tell me a joke."}],
stream=True,
caching=True)

View file

@ -18,12 +18,12 @@ litellm.cache = Cache()
# Make completion calls
response1 = completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Tell me a joke."}]
caching=True
)
response2 = completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Tell me a joke."}],
caching=True
)
@ -55,14 +55,14 @@ litellm.cache = Cache()
# Make completion calls
response1 = completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Tell me a joke."}],
stream=True,
caching=True)
for chunk in response1:
print(chunk)
response2 = completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Tell me a joke."}],
stream=True,
caching=True)

View file

@ -50,9 +50,9 @@ a. Setup config.yaml
```yaml
model_list:
- model_name: gpt-3.5-turbo
- model_name: gpt-4o
litellm_params:
model: openai/gpt-3.5-turbo
model: openai/gpt-4o
api_key: os.environ/OPENAI_API_KEY
- model_name: anthropic-claude
litellm_params:
@ -80,7 +80,7 @@ curl -L -X POST 'http://0.0.0.0:4000/chat/completions' \
-H 'Content-Type: application/json' \
-H 'Authorization: Bearer sk-1234' \
-d '{
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"messages": [
{
"role": "system",

View file

@ -17,7 +17,7 @@ os.environ["COHERE_API_KEY"] = "cohere key"
messages = [{ "content": "Hello, how are you?","role": "user"}]
# openai call
response = completion(model="gpt-3.5-turbo", messages=messages)
response = completion(model="gpt-4o", messages=messages)
# cohere call
response = completion("command-nightly", messages)
@ -59,7 +59,7 @@ os.environ["COHERE_API_KEY"] = "cohere key"
messages = [{ "content": "Hello, how are you?","role": "user"}]
# openai call
response = completion(model="gpt-3.5-turbo", messages=messages, logger_fn=my_custom_logging_fn)
response = completion(model="gpt-4o", messages=messages, logger_fn=my_custom_logging_fn)
# cohere call
response = completion("command-nightly", messages, logger_fn=my_custom_logging_fn)

View file

@ -11,9 +11,9 @@ import TabItem from '@theme/TabItem';
|---------------------------|-----------------------------------------------------------------|
| fine tuned `gpt-4-0613` | `response = completion(model="ft:gpt-4-0613", messages=messages)` |
| fine tuned `gpt-4o-2024-05-13` | `response = completion(model="ft:gpt-4o-2024-05-13", messages=messages)` |
| fine tuned `gpt-3.5-turbo-0125` | `response = completion(model="ft:gpt-3.5-turbo-0125", messages=messages)` |
| fine tuned `gpt-3.5-turbo-1106` | `response = completion(model="ft:gpt-3.5-turbo-1106", messages=messages)` |
| fine tuned `gpt-3.5-turbo-0613` | `response = completion(model="ft:gpt-3.5-turbo-0613", messages=messages)` |
| fine tuned `gpt-4o-0125` | `response = completion(model="ft:gpt-4o-0125", messages=messages)` |
| fine tuned `gpt-4o-1106` | `response = completion(model="ft:gpt-4o-1106", messages=messages)` |
| fine tuned `gpt-4o-0613` | `response = completion(model="ft:gpt-4o-0613", messages=messages)` |
## Vertex AI

View file

@ -41,7 +41,7 @@ model_list:
model: anthropic/claude-3-sonnet-20240229
api_key: os.environ/ANTHROPIC_API_KEY
- model_name: gpt-3.5-turbo
- model_name: gpt-4o
litellm_params:
model: azure/gpt-35-turbo
api_key: os.environ/AZURE_API_KEY
@ -726,7 +726,7 @@ agents['reviewer'] = client.create_agent(
name="reviewer",
system="You are an editor. Review and improve content quality.",
llm_config=LLMConfig(
model="openai/gpt-3.5-turbo",
model="openai/gpt-4o",
model_endpoint_type="openai"
)
)
@ -795,7 +795,7 @@ print(article)
1. **Model Selection**: Choose models based on task requirements:
- Use `openai/gpt-4` for complex reasoning
- Use `anthropic/claude-3-sonnet-20240229` for analysis
- Use `openai/gpt-3.5-turbo` for cost-effective simple tasks
- Use `openai/gpt-4o` for cost-effective simple tasks
2. **Error Handling**: Implement robust error handling with retries:
```python
@ -813,7 +813,7 @@ print(article)
except Exception as e:
print(f"LLM call failed: {e}")
# Implement fallback logic
return completion(model="openai/gpt-3.5-turbo", **kwargs)
return completion(model="openai/gpt-4o", **kwargs)
```
3. **Cost Management**:
@ -882,7 +882,7 @@ print("Anthropic Key:", os.environ.get("ANTHROPIC_API_KEY", "Not set"))
# Test direct LiteLLM call
try:
response = litellm.completion(
model="openai/gpt-3.5-turbo",
model="openai/gpt-4o",
messages=[{"role": "user", "content": "Hello"}]
)
print("LiteLLM working:", response.choices[0].message.content)

View file

@ -24,7 +24,7 @@ from langchain_core.prompts import (
from langchain_core.messages import AIMessage, HumanMessage, SystemMessage
os.environ['OPENAI_API_KEY'] = ""
chat = ChatLiteLLM(model="gpt-3.5-turbo")
chat = ChatLiteLLM(model="gpt-4o")
messages = [
HumanMessage(
content="what model are you"
@ -455,7 +455,7 @@ tag_routing:
- tags: ["premium", "high-priority"]
models: ["gpt-4o", "claude-3-opus"]
- tags: ["standard"]
models: ["gpt-3.5-turbo", "claude-3-haiku"]
models: ["gpt-4o", "claude-3-haiku"]
```
### Monitoring and Analytics

View file

@ -100,7 +100,7 @@ export AZURE_API_KEY=""
```yaml
model_list:
- model_name: gpt-3.5-turbo
- model_name: gpt-4o
litellm_params:
model: azure/chatgpt-v-2
api_base: https://openai-gpt-4-test-v-1.openai.azure.com/
@ -117,7 +117,7 @@ model_list:
curl --location 'http://0.0.0.0:4000/chat/completions' \
--header 'Content-Type: application/json' \
--data ' {
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"messages": [
{
"role": "user",
@ -137,7 +137,7 @@ client = openai.OpenAI(
base_url="http://0.0.0.0:4000"
)
response = client.chat.completions.create(model="gpt-3.5-turbo", messages = [
response = client.chat.completions.create(model="gpt-4o", messages = [
{
"role": "user",
"content": "this is a test request, write a short poem"
@ -161,7 +161,7 @@ from langchain.schema import HumanMessage, SystemMessage
chat = ChatOpenAI(
openai_api_base="http://0.0.0.0:4000", # set openai_api_base to the LiteLLM Proxy
model = "gpt-3.5-turbo",
model = "gpt-4o",
temperature=0.1
)
@ -224,11 +224,11 @@ model_list:
| gpt-4-32k-0613 | `completion('azure/<your deployment name>', messages)` |
| gpt-4-1106-preview | `completion('azure/<your deployment name>', messages)` |
| gpt-4-0125-preview | `completion('azure/<your deployment name>', messages)` |
| gpt-3.5-turbo | `completion('azure/<your deployment name>', messages)` |
| gpt-3.5-turbo-0301 | `completion('azure/<your deployment name>', messages)` |
| gpt-3.5-turbo-0613 | `completion('azure/<your deployment name>', messages)` |
| gpt-3.5-turbo-16k | `completion('azure/<your deployment name>', messages)` |
| gpt-3.5-turbo-16k-0613 | `completion('azure/<your deployment name>', messages)`
| gpt-4o | `completion('azure/<your deployment name>', messages)` |
| gpt-4o-0301 | `completion('azure/<your deployment name>', messages)` |
| gpt-4o-0613 | `completion('azure/<your deployment name>', messages)` |
| gpt-4o-16k | `completion('azure/<your deployment name>', messages)` |
| gpt-4o-16k-0613 | `completion('azure/<your deployment name>', messages)`
## Azure OpenAI Vision Models
| Model Name | Function Call |
@ -524,8 +524,8 @@ Use `model="azure_text/<your-deployment>"`
| Model Name | Function Call |
|---------------------|----------------------------------------------------|
| gpt-3.5-turbo-instruct | `response = completion(model="azure_text/<your deployment name>", messages=messages)` |
| gpt-3.5-turbo-instruct-0914 | `response = completion(model="azure_text/<your deployment name>", messages=messages)` |
| gpt-4o-instruct | `response = completion(model="azure_text/<your deployment name>", messages=messages)` |
| gpt-4o-instruct-0914 | `response = completion(model="azure_text/<your deployment name>", messages=messages)` |
```python
@ -604,7 +604,7 @@ response = litellm.completion(
```yaml
model_list:
- model_name: gpt-3.5-turbo
- model_name: gpt-4o
litellm_params:
model: azure/chatgpt-v-2
api_base: https://openai-gpt-4-test-v-1.openai.azure.com/
@ -620,7 +620,7 @@ model_list:
Here is an example of setting up `tenant_id`, `client_id`, `client_secret` in your litellm proxy `config.yaml`
```yaml
model_list:
- model_name: gpt-3.5-turbo
- model_name: gpt-4o
litellm_params:
model: azure/chatgpt-v-2
api_base: https://openai-gpt-4-test-v-1.openai.azure.com/
@ -637,7 +637,7 @@ Test it
curl --location 'http://0.0.0.0:4000/chat/completions' \
--header 'Content-Type: application/json' \
--data ' {
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"messages": [
{
"role": "user",
@ -657,7 +657,7 @@ Example video of using `tenant_id`, `client_id`, `client_secret` with LiteLLM Pr
Here is an example of setting up `client_id`, `azure_username`, `azure_password` in your litellm proxy `config.yaml`
```yaml
model_list:
- model_name: gpt-3.5-turbo
- model_name: gpt-4o
litellm_params:
model: azure/chatgpt-v-2
api_base: https://openai-gpt-4-test-v-1.openai.azure.com/
@ -674,7 +674,7 @@ Test it
curl --location 'http://0.0.0.0:4000/chat/completions' \
--header 'Content-Type: application/json' \
--data ' {
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"messages": [
{
"role": "user",
@ -746,7 +746,7 @@ export AZURE_CLIENT_SECRET=""
```yaml
model_list:
- model_name: gpt-3.5-turbo
- model_name: gpt-4o
litellm_params:
model: azure/your-deployment-name
api_base: https://openai-gpt-4-test-v-1.openai.azure.com/
@ -761,7 +761,7 @@ Perfect for AKS clusters, Azure VMs, or other managed environments where Azure a
```yaml
model_list:
- model_name: gpt-3.5-turbo
- model_name: gpt-4o
litellm_params:
model: azure/your-deployment-name
api_base: https://openai-gpt-4-test-v-1.openai.azure.com/
@ -776,7 +776,7 @@ If you're authenticated via `az login`, no additional configuration needed:
```yaml
model_list:
- model_name: gpt-3.5-turbo
- model_name: gpt-4o
litellm_params:
model: azure/your-deployment-name
api_base: https://openai-gpt-4-test-v-1.openai.azure.com/
@ -1150,7 +1150,7 @@ pip install litellm
from litellm import Router
model_list = [{ # list of model deployments
"model_name": "gpt-3.5-turbo", # openai model name
"model_name": "gpt-4o", # openai model name
"litellm_params": { # params for litellm completion/embedding call
"model": "azure/chatgpt-v-2",
"api_key": os.getenv("AZURE_API_KEY"),
@ -1160,7 +1160,7 @@ model_list = [{ # list of model deployments
"tpm": 240000,
"rpm": 1800
}, {
"model_name": "gpt-3.5-turbo", # openai model name
"model_name": "gpt-4o", # openai model name
"litellm_params": { # params for litellm completion/embedding call
"model": "azure/chatgpt-functioncalling",
"api_key": os.getenv("AZURE_API_KEY"),
@ -1170,9 +1170,9 @@ model_list = [{ # list of model deployments
"tpm": 240000,
"rpm": 1800
}, {
"model_name": "gpt-3.5-turbo", # openai model name
"model_name": "gpt-4o", # openai model name
"litellm_params": { # params for litellm completion/embedding call
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"api_key": os.getenv("OPENAI_API_KEY"),
},
"tpm": 1000000,
@ -1182,7 +1182,7 @@ model_list = [{ # list of model deployments
router = Router(model_list=model_list)
# openai.chat.completions.create replacement
response = router.completion(model="gpt-3.5-turbo",
response = router.completion(model="gpt-4o",
messages=[{"role": "user", "content": "Hey, how's it going?"}]
print(response)

View file

@ -31,7 +31,7 @@ from litellm import CustomLLM, completion, get_llm_provider
class MyCustomLLM(CustomLLM):
def completion(self, *args, **kwargs) -> litellm.ModelResponse:
return litellm.completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Hello world"}],
mock_response="Hi!",
) # type: ignore
@ -62,14 +62,14 @@ from litellm import CustomLLM, completion, get_llm_provider
class MyCustomLLM(CustomLLM):
def completion(self, *args, **kwargs) -> litellm.ModelResponse:
return litellm.completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Hello world"}],
mock_response="Hi!",
) # type: ignore
async def acompletion(self, *args, **kwargs) -> litellm.ModelResponse:
return litellm.completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Hello world"}],
mock_response="Hi!",
) # type: ignore
@ -135,7 +135,7 @@ Expected Response
}
],
"created": 1721955063,
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"object": "chat.completion",
"system_fingerprint": null,
"usage": {
@ -356,7 +356,7 @@ from litellm import CustomLLM, completion, get_llm_provider
class MyCustomLLM(CustomLLM):
async def acompletion(self, *args, **kwargs) -> litellm.ModelResponse:
return litellm.completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Hello world"}],
mock_response="Hi!",
) # type: ignore
@ -420,7 +420,7 @@ Expected Response
"id": "chatcmpl-Bm4qEp4h4vCe7Zi4Gud1MAxTWgibO",
"type": "message",
"role": "assistant",
"model": "gpt-3.5-turbo-0125",
"model": "gpt-4o-0125",
"stop_sequence": null,
"usage": {
"input_tokens": 18,
@ -455,7 +455,7 @@ class MyCustomLLM(CustomLLM):
def completion(self, *args, **kwargs) -> litellm.ModelResponse:
assert kwargs["optional_params"] == {"my_custom_param": "my-custom-param"} # 👈 CHECK HERE
return litellm.completion(
model="gpt-3.5-turbo",
model="gpt-4o",
messages=[{"role": "user", "content": "Hello world"}],
mock_response="Hi!",
) # type: ignore

View file

@ -51,8 +51,8 @@ This approach provides better flexibility for managing configurations across dif
| Model Name | Function Call |
|---------------------------|-----------------------------------------------------|
| openrouter/openai/gpt-3.5-turbo | `completion('openrouter/openai/gpt-3.5-turbo', messages)` | `os.environ['OR_SITE_URL']`,`os.environ['OR_APP_NAME']`,`os.environ['OPENROUTER_API_KEY']` |
| openrouter/openai/gpt-3.5-turbo-16k | `completion('openrouter/openai/gpt-3.5-turbo-16k', messages)` | `os.environ['OR_SITE_URL']`,`os.environ['OR_APP_NAME']`,`os.environ['OPENROUTER_API_KEY']` |
| openrouter/openai/gpt-4o | `completion('openrouter/openai/gpt-4o', messages)` | `os.environ['OR_SITE_URL']`,`os.environ['OR_APP_NAME']`,`os.environ['OPENROUTER_API_KEY']` |
| openrouter/openai/gpt-4o-16k | `completion('openrouter/openai/gpt-4o-16k', messages)` | `os.environ['OR_SITE_URL']`,`os.environ['OR_APP_NAME']`,`os.environ['OPENROUTER_API_KEY']` |
| openrouter/openai/gpt-4 | `completion('openrouter/openai/gpt-4', messages)` | `os.environ['OR_SITE_URL']`,`os.environ['OR_APP_NAME']`,`os.environ['OPENROUTER_API_KEY']` |
| openrouter/openai/gpt-4-32k | `completion('openrouter/openai/gpt-4-32k', messages)` | `os.environ['OR_SITE_URL']`,`os.environ['OR_APP_NAME']`,`os.environ['OPENROUTER_API_KEY']` |
| openrouter/anthropic/claude-2 | `completion('openrouter/anthropic/claude-2', messages)` | `os.environ['OR_SITE_URL']`,`os.environ['OR_APP_NAME']`,`os.environ['OPENROUTER_API_KEY']` |

View file

@ -21,7 +21,7 @@ os.environ["OPENAI_API_KEY"] = "your-api-key"
# openai call
response = completion(
model = "gpt-3.5-turbo-instruct",
model = "gpt-4o-instruct",
messages=[{ "content": "Hello, how are you?","role": "user"}]
)
```
@ -43,20 +43,20 @@ export OPENAI_API_KEY=""
```yaml
model_list:
- model_name: gpt-3.5-turbo
- model_name: gpt-4o
litellm_params:
model: openai/gpt-3.5-turbo # The `openai/` prefix will call openai.chat.completions.create
model: openai/gpt-4o # The `openai/` prefix will call openai.chat.completions.create
api_key: os.environ/OPENAI_API_KEY
- model_name: gpt-3.5-turbo-instruct
- model_name: gpt-4o-instruct
litellm_params:
model: text-completion-openai/gpt-3.5-turbo-instruct # The `text-completion-openai/` prefix will call openai.completions.create
model: text-completion-openai/gpt-4o-instruct # The `text-completion-openai/` prefix will call openai.completions.create
api_key: os.environ/OPENAI_API_KEY
```
</TabItem>
<TabItem value="config-*" label="config.yaml - proxy all OpenAI models">
Use this to add all openai models with one API Key. **WARNING: This will not do any load balancing**
This means requests to `gpt-4`, `gpt-3.5-turbo` , `gpt-4-turbo-preview` will all go through this route
This means requests to `gpt-4`, `gpt-4o` , `gpt-4-turbo-preview` will all go through this route
```yaml
model_list:
@ -69,7 +69,7 @@ model_list:
<TabItem value="cli" label="CLI">
```bash
$ litellm --model gpt-3.5-turbo-instruct
$ litellm --model gpt-4o-instruct
# Server running on http://0.0.0.0:4000
```
@ -87,7 +87,7 @@ $ litellm --model gpt-3.5-turbo-instruct
curl --location 'http://0.0.0.0:4000/chat/completions' \
--header 'Content-Type: application/json' \
--data ' {
"model": "gpt-3.5-turbo-instruct",
"model": "gpt-4o-instruct",
"messages": [
{
"role": "user",
@ -108,7 +108,7 @@ client = openai.OpenAI(
)
# request sent to model set on litellm proxy, `litellm --model`
response = client.chat.completions.create(model="gpt-3.5-turbo-instruct", messages = [
response = client.chat.completions.create(model="gpt-4o-instruct", messages = [
{
"role": "user",
"content": "this is a test request, write a short poem"
@ -132,7 +132,7 @@ from langchain.schema import HumanMessage, SystemMessage
chat = ChatOpenAI(
openai_api_base="http://0.0.0.0:4000", # set openai_api_base to the LiteLLM Proxy
model = "gpt-3.5-turbo-instruct",
model = "gpt-4o-instruct",
temperature=0.1
)
@ -156,8 +156,8 @@ print(response)
| Model Name | Function Call |
|---------------------|----------------------------------------------------|
| gpt-3.5-turbo-instruct | `response = completion(model="gpt-3.5-turbo-instruct", messages=messages)` |
| gpt-3.5-turbo-instruct-0914 | `response = completion(model="gpt-3.5-turbo-instruct-0914", messages=messages)` |
| gpt-4o-instruct | `response = completion(model="gpt-4o-instruct", messages=messages)` |
| gpt-4o-instruct-0914 | `response = completion(model="gpt-4o-instruct-0914", messages=messages)` |
| text-davinci-003 | `response = completion(model="text-davinci-003", messages=messages)` |
| ada-001 | `response = completion(model="ada-001", messages=messages)` |
| curie-001 | `response = completion(model="curie-001", messages=messages)` |