mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
docs: replace gpt-3.5-turbo with gpt-4o in provider and guide examples
Update model strings across OpenRouter, Azure, caching, LangChain, Letta, and related docs using StrReplace (replace_all) for consistency with gpt-4o. Co-authored-by: Krish Dholakia <krrishdholakia@gmail.com>
This commit is contained in:
parent
2a2386071d
commit
c4c2afc579
14 changed files with 93 additions and 93 deletions
|
|
@ -321,7 +321,7 @@ curl -X POST 'http://localhost:4000/{my_endpoint}' \
|
|||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer your-api-key' \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [{"role": "user", "content": "Hello"}],
|
||||
"guardrails": ["test"]
|
||||
}'
|
||||
|
|
|
|||
|
|
@ -131,9 +131,9 @@ Add to `config.yaml`:
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: openai/gpt-3.5-turbo
|
||||
model: openai/gpt-4o
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
prompts:
|
||||
|
|
|
|||
|
|
@ -39,11 +39,11 @@ litellm.cache = Cache(type="redis", host=<host>, port=<port>, password=<password
|
|||
|
||||
# Make completion calls
|
||||
response1 = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Tell me a joke."}]
|
||||
)
|
||||
response2 = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Tell me a joke."}]
|
||||
)
|
||||
|
||||
|
|
@ -77,11 +77,11 @@ litellm.cache = RedisClusterCache(
|
|||
|
||||
# Make completion calls
|
||||
response1 = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Tell me a joke."}]
|
||||
)
|
||||
response2 = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Tell me a joke."}]
|
||||
)
|
||||
|
||||
|
|
@ -132,11 +132,11 @@ from litellm.caching.caching import Cache
|
|||
litellm.cache = Cache(type="gcs", gcs_bucket_name="my-cache-bucket", gcs_path_service_account="/path/to/service_account.json")
|
||||
|
||||
response1 = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Tell me a joke."}]
|
||||
)
|
||||
response2 = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Tell me a joke."}]
|
||||
)
|
||||
|
||||
|
|
@ -170,11 +170,11 @@ litellm.cache = Cache(type="s3", s3_bucket_name="cache-bucket-litellm", s3_regio
|
|||
|
||||
# Make completion calls
|
||||
response1 = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Tell me a joke."}]
|
||||
)
|
||||
response2 = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Tell me a joke."}]
|
||||
)
|
||||
|
||||
|
|
@ -201,11 +201,11 @@ litellm.cache = Cache(type="azure-blob", azure_account_url="https://example.blob
|
|||
|
||||
# Make completion calls
|
||||
response1 = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Tell me a joke."}]
|
||||
)
|
||||
response2 = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Tell me a joke."}]
|
||||
)
|
||||
|
||||
|
|
@ -244,7 +244,7 @@ litellm.cache = Cache(
|
|||
redis_semantic_cache_embedding_model="text-embedding-ada-002", # this model is passed to litellm.embedding(), any litellm.embedding() model is supported here
|
||||
)
|
||||
response1 = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -258,7 +258,7 @@ print(f"response1: {response1}")
|
|||
random_number = random.randint(1, 100000)
|
||||
|
||||
response2 = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -301,7 +301,7 @@ litellm.cache = Cache(
|
|||
)
|
||||
|
||||
response1 = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -315,7 +315,7 @@ print(f"response1: {response1}")
|
|||
random_number = random.randint(1, 100000)
|
||||
|
||||
response2 = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -343,12 +343,12 @@ litellm.cache = Cache()
|
|||
|
||||
# Make completion calls
|
||||
response1 = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Tell me a joke."}],
|
||||
caching=True
|
||||
)
|
||||
response2 = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Tell me a joke."}],
|
||||
caching=True
|
||||
)
|
||||
|
|
@ -379,12 +379,12 @@ litellm.cache = Cache(type="disk")
|
|||
|
||||
# Make completion calls
|
||||
response1 = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Tell me a joke."}],
|
||||
caching=True
|
||||
)
|
||||
response2 = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Tell me a joke."}],
|
||||
caching=True
|
||||
)
|
||||
|
|
@ -416,7 +416,7 @@ Example usage `no-cache` - When `True`, Will not return a cached response
|
|||
|
||||
```python
|
||||
response = litellm.completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -435,7 +435,7 @@ Example usage `no-store` - When `True`, Will not cache the response.
|
|||
|
||||
```python
|
||||
response = litellm.completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -453,7 +453,7 @@ Example usage `ttl` - cache the response for 10 seconds
|
|||
|
||||
```python
|
||||
response = litellm.completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -471,7 +471,7 @@ Example usage `s-maxage` - Will only accept cached responses for 60 seconds
|
|||
|
||||
```python
|
||||
response = litellm.completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
|
|
|
|||
|
|
@ -11,13 +11,13 @@ litellm.cache = Cache(type="hosted") # init cache to use api.litellm.ai
|
|||
|
||||
# Make completion calls
|
||||
response1 = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Tell me a joke."}]
|
||||
caching=True
|
||||
)
|
||||
|
||||
response2 = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Tell me a joke."}],
|
||||
caching=True
|
||||
)
|
||||
|
|
@ -59,7 +59,7 @@ litellm.cache = Cache(type="hosted")
|
|||
|
||||
# Make completion calls
|
||||
response1 = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Tell me a joke."}],
|
||||
stream=True,
|
||||
caching=True)
|
||||
|
|
@ -69,7 +69,7 @@ for chunk in response1:
|
|||
time.sleep(1) # cache is updated asynchronously
|
||||
|
||||
response2 = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Tell me a joke."}],
|
||||
stream=True,
|
||||
caching=True)
|
||||
|
|
|
|||
|
|
@ -18,12 +18,12 @@ litellm.cache = Cache()
|
|||
|
||||
# Make completion calls
|
||||
response1 = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Tell me a joke."}]
|
||||
caching=True
|
||||
)
|
||||
response2 = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Tell me a joke."}],
|
||||
caching=True
|
||||
)
|
||||
|
|
@ -55,14 +55,14 @@ litellm.cache = Cache()
|
|||
|
||||
# Make completion calls
|
||||
response1 = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Tell me a joke."}],
|
||||
stream=True,
|
||||
caching=True)
|
||||
for chunk in response1:
|
||||
print(chunk)
|
||||
response2 = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Tell me a joke."}],
|
||||
stream=True,
|
||||
caching=True)
|
||||
|
|
|
|||
|
|
@ -50,9 +50,9 @@ a. Setup config.yaml
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: openai/gpt-3.5-turbo
|
||||
model: openai/gpt-4o
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
- model_name: anthropic-claude
|
||||
litellm_params:
|
||||
|
|
@ -80,7 +80,7 @@ curl -L -X POST 'http://0.0.0.0:4000/chat/completions' \
|
|||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "system",
|
||||
|
|
|
|||
|
|
@ -17,7 +17,7 @@ os.environ["COHERE_API_KEY"] = "cohere key"
|
|||
messages = [{ "content": "Hello, how are you?","role": "user"}]
|
||||
|
||||
# openai call
|
||||
response = completion(model="gpt-3.5-turbo", messages=messages)
|
||||
response = completion(model="gpt-4o", messages=messages)
|
||||
|
||||
# cohere call
|
||||
response = completion("command-nightly", messages)
|
||||
|
|
@ -59,7 +59,7 @@ os.environ["COHERE_API_KEY"] = "cohere key"
|
|||
messages = [{ "content": "Hello, how are you?","role": "user"}]
|
||||
|
||||
# openai call
|
||||
response = completion(model="gpt-3.5-turbo", messages=messages, logger_fn=my_custom_logging_fn)
|
||||
response = completion(model="gpt-4o", messages=messages, logger_fn=my_custom_logging_fn)
|
||||
|
||||
# cohere call
|
||||
response = completion("command-nightly", messages, logger_fn=my_custom_logging_fn)
|
||||
|
|
|
|||
|
|
@ -11,9 +11,9 @@ import TabItem from '@theme/TabItem';
|
|||
|---------------------------|-----------------------------------------------------------------|
|
||||
| fine tuned `gpt-4-0613` | `response = completion(model="ft:gpt-4-0613", messages=messages)` |
|
||||
| fine tuned `gpt-4o-2024-05-13` | `response = completion(model="ft:gpt-4o-2024-05-13", messages=messages)` |
|
||||
| fine tuned `gpt-3.5-turbo-0125` | `response = completion(model="ft:gpt-3.5-turbo-0125", messages=messages)` |
|
||||
| fine tuned `gpt-3.5-turbo-1106` | `response = completion(model="ft:gpt-3.5-turbo-1106", messages=messages)` |
|
||||
| fine tuned `gpt-3.5-turbo-0613` | `response = completion(model="ft:gpt-3.5-turbo-0613", messages=messages)` |
|
||||
| fine tuned `gpt-4o-0125` | `response = completion(model="ft:gpt-4o-0125", messages=messages)` |
|
||||
| fine tuned `gpt-4o-1106` | `response = completion(model="ft:gpt-4o-1106", messages=messages)` |
|
||||
| fine tuned `gpt-4o-0613` | `response = completion(model="ft:gpt-4o-0613", messages=messages)` |
|
||||
|
||||
|
||||
## Vertex AI
|
||||
|
|
|
|||
|
|
@ -41,7 +41,7 @@ model_list:
|
|||
model: anthropic/claude-3-sonnet-20240229
|
||||
api_key: os.environ/ANTHROPIC_API_KEY
|
||||
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: azure/gpt-35-turbo
|
||||
api_key: os.environ/AZURE_API_KEY
|
||||
|
|
@ -726,7 +726,7 @@ agents['reviewer'] = client.create_agent(
|
|||
name="reviewer",
|
||||
system="You are an editor. Review and improve content quality.",
|
||||
llm_config=LLMConfig(
|
||||
model="openai/gpt-3.5-turbo",
|
||||
model="openai/gpt-4o",
|
||||
model_endpoint_type="openai"
|
||||
)
|
||||
)
|
||||
|
|
@ -795,7 +795,7 @@ print(article)
|
|||
1. **Model Selection**: Choose models based on task requirements:
|
||||
- Use `openai/gpt-4` for complex reasoning
|
||||
- Use `anthropic/claude-3-sonnet-20240229` for analysis
|
||||
- Use `openai/gpt-3.5-turbo` for cost-effective simple tasks
|
||||
- Use `openai/gpt-4o` for cost-effective simple tasks
|
||||
|
||||
2. **Error Handling**: Implement robust error handling with retries:
|
||||
```python
|
||||
|
|
@ -813,7 +813,7 @@ print(article)
|
|||
except Exception as e:
|
||||
print(f"LLM call failed: {e}")
|
||||
# Implement fallback logic
|
||||
return completion(model="openai/gpt-3.5-turbo", **kwargs)
|
||||
return completion(model="openai/gpt-4o", **kwargs)
|
||||
```
|
||||
|
||||
3. **Cost Management**:
|
||||
|
|
@ -882,7 +882,7 @@ print("Anthropic Key:", os.environ.get("ANTHROPIC_API_KEY", "Not set"))
|
|||
# Test direct LiteLLM call
|
||||
try:
|
||||
response = litellm.completion(
|
||||
model="openai/gpt-3.5-turbo",
|
||||
model="openai/gpt-4o",
|
||||
messages=[{"role": "user", "content": "Hello"}]
|
||||
)
|
||||
print("LiteLLM working:", response.choices[0].message.content)
|
||||
|
|
|
|||
|
|
@ -24,7 +24,7 @@ from langchain_core.prompts import (
|
|||
from langchain_core.messages import AIMessage, HumanMessage, SystemMessage
|
||||
|
||||
os.environ['OPENAI_API_KEY'] = ""
|
||||
chat = ChatLiteLLM(model="gpt-3.5-turbo")
|
||||
chat = ChatLiteLLM(model="gpt-4o")
|
||||
messages = [
|
||||
HumanMessage(
|
||||
content="what model are you"
|
||||
|
|
@ -455,7 +455,7 @@ tag_routing:
|
|||
- tags: ["premium", "high-priority"]
|
||||
models: ["gpt-4o", "claude-3-opus"]
|
||||
- tags: ["standard"]
|
||||
models: ["gpt-3.5-turbo", "claude-3-haiku"]
|
||||
models: ["gpt-4o", "claude-3-haiku"]
|
||||
```
|
||||
|
||||
### Monitoring and Analytics
|
||||
|
|
|
|||
|
|
@ -100,7 +100,7 @@ export AZURE_API_KEY=""
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: azure/chatgpt-v-2
|
||||
api_base: https://openai-gpt-4-test-v-1.openai.azure.com/
|
||||
|
|
@ -117,7 +117,7 @@ model_list:
|
|||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data ' {
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -137,7 +137,7 @@ client = openai.OpenAI(
|
|||
base_url="http://0.0.0.0:4000"
|
||||
)
|
||||
|
||||
response = client.chat.completions.create(model="gpt-3.5-turbo", messages = [
|
||||
response = client.chat.completions.create(model="gpt-4o", messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "this is a test request, write a short poem"
|
||||
|
|
@ -161,7 +161,7 @@ from langchain.schema import HumanMessage, SystemMessage
|
|||
|
||||
chat = ChatOpenAI(
|
||||
openai_api_base="http://0.0.0.0:4000", # set openai_api_base to the LiteLLM Proxy
|
||||
model = "gpt-3.5-turbo",
|
||||
model = "gpt-4o",
|
||||
temperature=0.1
|
||||
)
|
||||
|
||||
|
|
@ -224,11 +224,11 @@ model_list:
|
|||
| gpt-4-32k-0613 | `completion('azure/<your deployment name>', messages)` |
|
||||
| gpt-4-1106-preview | `completion('azure/<your deployment name>', messages)` |
|
||||
| gpt-4-0125-preview | `completion('azure/<your deployment name>', messages)` |
|
||||
| gpt-3.5-turbo | `completion('azure/<your deployment name>', messages)` |
|
||||
| gpt-3.5-turbo-0301 | `completion('azure/<your deployment name>', messages)` |
|
||||
| gpt-3.5-turbo-0613 | `completion('azure/<your deployment name>', messages)` |
|
||||
| gpt-3.5-turbo-16k | `completion('azure/<your deployment name>', messages)` |
|
||||
| gpt-3.5-turbo-16k-0613 | `completion('azure/<your deployment name>', messages)`
|
||||
| gpt-4o | `completion('azure/<your deployment name>', messages)` |
|
||||
| gpt-4o-0301 | `completion('azure/<your deployment name>', messages)` |
|
||||
| gpt-4o-0613 | `completion('azure/<your deployment name>', messages)` |
|
||||
| gpt-4o-16k | `completion('azure/<your deployment name>', messages)` |
|
||||
| gpt-4o-16k-0613 | `completion('azure/<your deployment name>', messages)`
|
||||
|
||||
## Azure OpenAI Vision Models
|
||||
| Model Name | Function Call |
|
||||
|
|
@ -524,8 +524,8 @@ Use `model="azure_text/<your-deployment>"`
|
|||
|
||||
| Model Name | Function Call |
|
||||
|---------------------|----------------------------------------------------|
|
||||
| gpt-3.5-turbo-instruct | `response = completion(model="azure_text/<your deployment name>", messages=messages)` |
|
||||
| gpt-3.5-turbo-instruct-0914 | `response = completion(model="azure_text/<your deployment name>", messages=messages)` |
|
||||
| gpt-4o-instruct | `response = completion(model="azure_text/<your deployment name>", messages=messages)` |
|
||||
| gpt-4o-instruct-0914 | `response = completion(model="azure_text/<your deployment name>", messages=messages)` |
|
||||
|
||||
|
||||
```python
|
||||
|
|
@ -604,7 +604,7 @@ response = litellm.completion(
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: azure/chatgpt-v-2
|
||||
api_base: https://openai-gpt-4-test-v-1.openai.azure.com/
|
||||
|
|
@ -620,7 +620,7 @@ model_list:
|
|||
Here is an example of setting up `tenant_id`, `client_id`, `client_secret` in your litellm proxy `config.yaml`
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: azure/chatgpt-v-2
|
||||
api_base: https://openai-gpt-4-test-v-1.openai.azure.com/
|
||||
|
|
@ -637,7 +637,7 @@ Test it
|
|||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data ' {
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -657,7 +657,7 @@ Example video of using `tenant_id`, `client_id`, `client_secret` with LiteLLM Pr
|
|||
Here is an example of setting up `client_id`, `azure_username`, `azure_password` in your litellm proxy `config.yaml`
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: azure/chatgpt-v-2
|
||||
api_base: https://openai-gpt-4-test-v-1.openai.azure.com/
|
||||
|
|
@ -674,7 +674,7 @@ Test it
|
|||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data ' {
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -746,7 +746,7 @@ export AZURE_CLIENT_SECRET=""
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: azure/your-deployment-name
|
||||
api_base: https://openai-gpt-4-test-v-1.openai.azure.com/
|
||||
|
|
@ -761,7 +761,7 @@ Perfect for AKS clusters, Azure VMs, or other managed environments where Azure a
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: azure/your-deployment-name
|
||||
api_base: https://openai-gpt-4-test-v-1.openai.azure.com/
|
||||
|
|
@ -776,7 +776,7 @@ If you're authenticated via `az login`, no additional configuration needed:
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: azure/your-deployment-name
|
||||
api_base: https://openai-gpt-4-test-v-1.openai.azure.com/
|
||||
|
|
@ -1150,7 +1150,7 @@ pip install litellm
|
|||
from litellm import Router
|
||||
|
||||
model_list = [{ # list of model deployments
|
||||
"model_name": "gpt-3.5-turbo", # openai model name
|
||||
"model_name": "gpt-4o", # openai model name
|
||||
"litellm_params": { # params for litellm completion/embedding call
|
||||
"model": "azure/chatgpt-v-2",
|
||||
"api_key": os.getenv("AZURE_API_KEY"),
|
||||
|
|
@ -1160,7 +1160,7 @@ model_list = [{ # list of model deployments
|
|||
"tpm": 240000,
|
||||
"rpm": 1800
|
||||
}, {
|
||||
"model_name": "gpt-3.5-turbo", # openai model name
|
||||
"model_name": "gpt-4o", # openai model name
|
||||
"litellm_params": { # params for litellm completion/embedding call
|
||||
"model": "azure/chatgpt-functioncalling",
|
||||
"api_key": os.getenv("AZURE_API_KEY"),
|
||||
|
|
@ -1170,9 +1170,9 @@ model_list = [{ # list of model deployments
|
|||
"tpm": 240000,
|
||||
"rpm": 1800
|
||||
}, {
|
||||
"model_name": "gpt-3.5-turbo", # openai model name
|
||||
"model_name": "gpt-4o", # openai model name
|
||||
"litellm_params": { # params for litellm completion/embedding call
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"api_key": os.getenv("OPENAI_API_KEY"),
|
||||
},
|
||||
"tpm": 1000000,
|
||||
|
|
@ -1182,7 +1182,7 @@ model_list = [{ # list of model deployments
|
|||
router = Router(model_list=model_list)
|
||||
|
||||
# openai.chat.completions.create replacement
|
||||
response = router.completion(model="gpt-3.5-turbo",
|
||||
response = router.completion(model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Hey, how's it going?"}]
|
||||
|
||||
print(response)
|
||||
|
|
|
|||
|
|
@ -31,7 +31,7 @@ from litellm import CustomLLM, completion, get_llm_provider
|
|||
class MyCustomLLM(CustomLLM):
|
||||
def completion(self, *args, **kwargs) -> litellm.ModelResponse:
|
||||
return litellm.completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Hello world"}],
|
||||
mock_response="Hi!",
|
||||
) # type: ignore
|
||||
|
|
@ -62,14 +62,14 @@ from litellm import CustomLLM, completion, get_llm_provider
|
|||
class MyCustomLLM(CustomLLM):
|
||||
def completion(self, *args, **kwargs) -> litellm.ModelResponse:
|
||||
return litellm.completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Hello world"}],
|
||||
mock_response="Hi!",
|
||||
) # type: ignore
|
||||
|
||||
async def acompletion(self, *args, **kwargs) -> litellm.ModelResponse:
|
||||
return litellm.completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Hello world"}],
|
||||
mock_response="Hi!",
|
||||
) # type: ignore
|
||||
|
|
@ -135,7 +135,7 @@ Expected Response
|
|||
}
|
||||
],
|
||||
"created": 1721955063,
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"object": "chat.completion",
|
||||
"system_fingerprint": null,
|
||||
"usage": {
|
||||
|
|
@ -356,7 +356,7 @@ from litellm import CustomLLM, completion, get_llm_provider
|
|||
class MyCustomLLM(CustomLLM):
|
||||
async def acompletion(self, *args, **kwargs) -> litellm.ModelResponse:
|
||||
return litellm.completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Hello world"}],
|
||||
mock_response="Hi!",
|
||||
) # type: ignore
|
||||
|
|
@ -420,7 +420,7 @@ Expected Response
|
|||
"id": "chatcmpl-Bm4qEp4h4vCe7Zi4Gud1MAxTWgibO",
|
||||
"type": "message",
|
||||
"role": "assistant",
|
||||
"model": "gpt-3.5-turbo-0125",
|
||||
"model": "gpt-4o-0125",
|
||||
"stop_sequence": null,
|
||||
"usage": {
|
||||
"input_tokens": 18,
|
||||
|
|
@ -455,7 +455,7 @@ class MyCustomLLM(CustomLLM):
|
|||
def completion(self, *args, **kwargs) -> litellm.ModelResponse:
|
||||
assert kwargs["optional_params"] == {"my_custom_param": "my-custom-param"} # 👈 CHECK HERE
|
||||
return litellm.completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Hello world"}],
|
||||
mock_response="Hi!",
|
||||
) # type: ignore
|
||||
|
|
|
|||
|
|
@ -51,8 +51,8 @@ This approach provides better flexibility for managing configurations across dif
|
|||
|
||||
| Model Name | Function Call |
|
||||
|---------------------------|-----------------------------------------------------|
|
||||
| openrouter/openai/gpt-3.5-turbo | `completion('openrouter/openai/gpt-3.5-turbo', messages)` | `os.environ['OR_SITE_URL']`,`os.environ['OR_APP_NAME']`,`os.environ['OPENROUTER_API_KEY']` |
|
||||
| openrouter/openai/gpt-3.5-turbo-16k | `completion('openrouter/openai/gpt-3.5-turbo-16k', messages)` | `os.environ['OR_SITE_URL']`,`os.environ['OR_APP_NAME']`,`os.environ['OPENROUTER_API_KEY']` |
|
||||
| openrouter/openai/gpt-4o | `completion('openrouter/openai/gpt-4o', messages)` | `os.environ['OR_SITE_URL']`,`os.environ['OR_APP_NAME']`,`os.environ['OPENROUTER_API_KEY']` |
|
||||
| openrouter/openai/gpt-4o-16k | `completion('openrouter/openai/gpt-4o-16k', messages)` | `os.environ['OR_SITE_URL']`,`os.environ['OR_APP_NAME']`,`os.environ['OPENROUTER_API_KEY']` |
|
||||
| openrouter/openai/gpt-4 | `completion('openrouter/openai/gpt-4', messages)` | `os.environ['OR_SITE_URL']`,`os.environ['OR_APP_NAME']`,`os.environ['OPENROUTER_API_KEY']` |
|
||||
| openrouter/openai/gpt-4-32k | `completion('openrouter/openai/gpt-4-32k', messages)` | `os.environ['OR_SITE_URL']`,`os.environ['OR_APP_NAME']`,`os.environ['OPENROUTER_API_KEY']` |
|
||||
| openrouter/anthropic/claude-2 | `completion('openrouter/anthropic/claude-2', messages)` | `os.environ['OR_SITE_URL']`,`os.environ['OR_APP_NAME']`,`os.environ['OPENROUTER_API_KEY']` |
|
||||
|
|
|
|||
|
|
@ -21,7 +21,7 @@ os.environ["OPENAI_API_KEY"] = "your-api-key"
|
|||
|
||||
# openai call
|
||||
response = completion(
|
||||
model = "gpt-3.5-turbo-instruct",
|
||||
model = "gpt-4o-instruct",
|
||||
messages=[{ "content": "Hello, how are you?","role": "user"}]
|
||||
)
|
||||
```
|
||||
|
|
@ -43,20 +43,20 @@ export OPENAI_API_KEY=""
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: openai/gpt-3.5-turbo # The `openai/` prefix will call openai.chat.completions.create
|
||||
model: openai/gpt-4o # The `openai/` prefix will call openai.chat.completions.create
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
- model_name: gpt-3.5-turbo-instruct
|
||||
- model_name: gpt-4o-instruct
|
||||
litellm_params:
|
||||
model: text-completion-openai/gpt-3.5-turbo-instruct # The `text-completion-openai/` prefix will call openai.completions.create
|
||||
model: text-completion-openai/gpt-4o-instruct # The `text-completion-openai/` prefix will call openai.completions.create
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
```
|
||||
</TabItem>
|
||||
<TabItem value="config-*" label="config.yaml - proxy all OpenAI models">
|
||||
|
||||
Use this to add all openai models with one API Key. **WARNING: This will not do any load balancing**
|
||||
This means requests to `gpt-4`, `gpt-3.5-turbo` , `gpt-4-turbo-preview` will all go through this route
|
||||
This means requests to `gpt-4`, `gpt-4o` , `gpt-4-turbo-preview` will all go through this route
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
|
|
@ -69,7 +69,7 @@ model_list:
|
|||
<TabItem value="cli" label="CLI">
|
||||
|
||||
```bash
|
||||
$ litellm --model gpt-3.5-turbo-instruct
|
||||
$ litellm --model gpt-4o-instruct
|
||||
|
||||
# Server running on http://0.0.0.0:4000
|
||||
```
|
||||
|
|
@ -87,7 +87,7 @@ $ litellm --model gpt-3.5-turbo-instruct
|
|||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data ' {
|
||||
"model": "gpt-3.5-turbo-instruct",
|
||||
"model": "gpt-4o-instruct",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -108,7 +108,7 @@ client = openai.OpenAI(
|
|||
)
|
||||
|
||||
# request sent to model set on litellm proxy, `litellm --model`
|
||||
response = client.chat.completions.create(model="gpt-3.5-turbo-instruct", messages = [
|
||||
response = client.chat.completions.create(model="gpt-4o-instruct", messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "this is a test request, write a short poem"
|
||||
|
|
@ -132,7 +132,7 @@ from langchain.schema import HumanMessage, SystemMessage
|
|||
|
||||
chat = ChatOpenAI(
|
||||
openai_api_base="http://0.0.0.0:4000", # set openai_api_base to the LiteLLM Proxy
|
||||
model = "gpt-3.5-turbo-instruct",
|
||||
model = "gpt-4o-instruct",
|
||||
temperature=0.1
|
||||
)
|
||||
|
||||
|
|
@ -156,8 +156,8 @@ print(response)
|
|||
|
||||
| Model Name | Function Call |
|
||||
|---------------------|----------------------------------------------------|
|
||||
| gpt-3.5-turbo-instruct | `response = completion(model="gpt-3.5-turbo-instruct", messages=messages)` |
|
||||
| gpt-3.5-turbo-instruct-0914 | `response = completion(model="gpt-3.5-turbo-instruct-0914", messages=messages)` |
|
||||
| gpt-4o-instruct | `response = completion(model="gpt-4o-instruct", messages=messages)` |
|
||||
| gpt-4o-instruct-0914 | `response = completion(model="gpt-4o-instruct-0914", messages=messages)` |
|
||||
| text-davinci-003 | `response = completion(model="text-davinci-003", messages=messages)` |
|
||||
| ada-001 | `response = completion(model="ada-001", messages=messages)` |
|
||||
| curie-001 | `response = completion(model="curie-001", messages=messages)` |
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue