mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
docs(proxy): replace gpt-3.5-turbo with gpt-4o in example configs
Update model name strings across proxy documentation pages for consistency with current OpenAI model guidance. Co-authored-by: Krish Dholakia <krrishdholakia@gmail.com>
This commit is contained in:
parent
bb741e73fc
commit
831bc1b6fd
15 changed files with 77 additions and 77 deletions
|
|
@ -42,7 +42,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
|||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"prompt_id": "simple_prompt",
|
||||
"prompt_variables": {
|
||||
"question": "Explain quantum computing"
|
||||
|
|
|
|||
|
|
@ -166,7 +166,7 @@ This page documents all command-line interface (CLI) arguments available for the
|
|||
- The model name to pass to LiteLLM.
|
||||
- **Usage:**
|
||||
```shell
|
||||
litellm --model gpt-3.5-turbo
|
||||
litellm --model gpt-4o
|
||||
```
|
||||
|
||||
### --alias
|
||||
|
|
@ -214,7 +214,7 @@ This page documents all command-line interface (CLI) arguments available for the
|
|||
- Save the model-specific config.
|
||||
- **Usage:**
|
||||
```shell
|
||||
litellm --model gpt-3.5-turbo --save
|
||||
litellm --model gpt-4o --save
|
||||
```
|
||||
|
||||
## Model Parameters
|
||||
|
|
|
|||
|
|
@ -182,7 +182,7 @@ async def user_api_key_auth(request: Request, api_key: str) -> UserAPIKeyAuth:
|
|||
soft_budget=800.0,
|
||||
tpm_limit=10000,
|
||||
rpm_limit=100,
|
||||
models=["gpt-4", "claude-3-sonnet", "gpt-3.5-turbo"],
|
||||
models=["gpt-4", "claude-3-sonnet", "gpt-4o"],
|
||||
allowed_routes=["/chat/completions", "/embeddings"],
|
||||
expires=datetime.now() + timedelta(days=30),
|
||||
metadata={"department": "engineering", "cost_center": "ai_ops"}
|
||||
|
|
@ -198,7 +198,7 @@ async def user_api_key_auth(request: Request, api_key: str) -> UserAPIKeyAuth:
|
|||
max_budget=100.0,
|
||||
tpm_limit=1000,
|
||||
rpm_limit=20,
|
||||
models=["gpt-3.5-turbo", "claude-3-haiku"],
|
||||
models=["gpt-4o", "claude-3-haiku"],
|
||||
team_member_tpm_limit=500, # Limit within team
|
||||
end_user_tpm_limit=100, # Per end-user limit
|
||||
metadata={"project": "chatbot_v2"}
|
||||
|
|
@ -218,7 +218,7 @@ e.g. if they're both in the same dir - `./config.yaml` and `./custom_auth.py`, t
|
|||
model_list:
|
||||
- model_name: "openai-model"
|
||||
litellm_params:
|
||||
model: "gpt-3.5-turbo"
|
||||
model: "gpt-4o"
|
||||
|
||||
litellm_settings:
|
||||
drop_params: True
|
||||
|
|
@ -286,7 +286,7 @@ Key change set `mode: auto`. This will check both litellm api key auth + custom
|
|||
model_list:
|
||||
- model_name: "openai-model"
|
||||
litellm_params:
|
||||
model: "gpt-3.5-turbo"
|
||||
model: "gpt-4o"
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
general_settings:
|
||||
|
|
|
|||
|
|
@ -37,13 +37,13 @@ Supported regions are 'eu' and 'us'.
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: azure/gpt-35-turbo # 👈 EU azure model
|
||||
api_base: https://my-endpoint-europe-berri-992.openai.azure.com/
|
||||
api_key: os.environ/AZURE_EUROPE_API_KEY
|
||||
region_name: "eu"
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: azure/chatgpt-v-2
|
||||
api_base: https://openai-gpt-4-test-v-1.openai.azure.com/
|
||||
|
|
@ -70,7 +70,7 @@ curl -X POST --location 'http://localhost:4000/chat/completions' \
|
|||
--header 'Content-Type: application/json' \
|
||||
--header 'Authorization: Bearer sk-1234' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
|
|||
|
|
@ -391,7 +391,7 @@ curl -X POST 'http://localhost:4000/chat/completions' \
|
|||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [{"role": "user", "content": "Hello"}],
|
||||
"user": "my-customer-id"
|
||||
}'
|
||||
|
|
@ -508,7 +508,7 @@ client = OpenAI(
|
|||
)
|
||||
|
||||
completion = client.chat.completions.create(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[
|
||||
{"role": "system", "content": "You are a helpful assistant."},
|
||||
{"role": "user", "content": "Hello!"}
|
||||
|
|
|
|||
|
|
@ -15,7 +15,7 @@ Dynamically allocate TPM/RPM quota to api keys, based on active keys in that min
|
|||
model_list:
|
||||
- model_name: my-fake-model
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
model: gpt-4o
|
||||
api_key: my-fake-key
|
||||
mock_response: hello-world
|
||||
tpm: 60
|
||||
|
|
@ -130,9 +130,9 @@ Priority reservation allocates a percentage of your model's total TPM/RPM to spe
|
|||
|
||||
```yaml showLineNumbers title="config.yaml"
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: "gpt-3.5-turbo"
|
||||
model: "gpt-4o"
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
rpm: 10 # Total model capacity
|
||||
|
||||
|
|
@ -262,7 +262,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
|||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-prod-key' \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [{"role": "user", "content": "Hello from prod"}]
|
||||
}'
|
||||
```
|
||||
|
|
@ -273,7 +273,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
|||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-dev-key' \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [{"role": "user", "content": "Hello from dev"}]
|
||||
}'
|
||||
```
|
||||
|
|
|
|||
|
|
@ -32,7 +32,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \ # 👈 ENDPOINT AUTOMATICA
|
|||
--header 'Content-Type: application/json' \
|
||||
--header 'Authorization: Bearer sk-1234' \ # 👈 YOUR PROXY KEY
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
|
|||
|
|
@ -103,7 +103,7 @@ curl --location 'http://localhost:4000/chat/completions' \
|
|||
--header 'Authorization: Bearer sk-5fmYeaUEbAMpwBNT-QpxyA' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -128,7 +128,7 @@ curl --location 'http://localhost:4000/chat/completions' \
|
|||
--header 'Authorization: Bearer sk-5fmYeaUEbAMpwBNT-QpxyA' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"user": "gm",
|
||||
"messages": [
|
||||
{
|
||||
|
|
@ -155,7 +155,7 @@ curl --location 'http://localhost:4000/chat/completions' \
|
|||
--header 'Authorization: Bearer sk-5fmYeaUEbAMpwBNT-QpxyA' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"user": "gm",
|
||||
"messages": [
|
||||
{
|
||||
|
|
@ -170,7 +170,7 @@ curl --location 'http://localhost:4000/chat/completions' \
|
|||
Expected Response
|
||||
|
||||
```shell
|
||||
{"id":"chatcmpl-9XALnHqkCBMBKrOx7Abg0hURHqYtY","choices":[{"finish_reason":"stop","index":0,"message":{"content":"Hello! How can I assist you today?","role":"assistant"}}],"created":1717691639,"model":"gpt-3.5-turbo-0125","object":"chat.completion","system_fingerprint":null,"usage":{"completion_tokens":9,"prompt_tokens":8,"total_tokens":17}}%
|
||||
{"id":"chatcmpl-9XALnHqkCBMBKrOx7Abg0hURHqYtY","choices":[{"finish_reason":"stop","index":0,"message":{"content":"Hello! How can I assist you today?","role":"assistant"}}],"created":1717691639,"model":"gpt-4o-0125","object":"chat.completion","system_fingerprint":null,"usage":{"completion_tokens":9,"prompt_tokens":8,"total_tokens":17}}%
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
|
@ -481,7 +481,7 @@ client = openai.OpenAI(
|
|||
|
||||
# request sent to model set on litellm proxy, `litellm --model`
|
||||
response = client.chat.completions.create(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -668,7 +668,7 @@ client = openai.OpenAI(
|
|||
|
||||
# request sent to model set on litellm proxy, `litellm --model`
|
||||
response = client.chat.completions.create(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -688,7 +688,7 @@ print(response)
|
|||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data ' {
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -748,7 +748,7 @@ litellm_settings:
|
|||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data ' {
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
|
|||
|
|
@ -20,7 +20,7 @@ Create or update fallbacks for a specific model.
|
|||
**Request Body:**
|
||||
```json
|
||||
{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"fallback_models": ["gpt-4", "claude-3-haiku"],
|
||||
"fallback_type": "general"
|
||||
}
|
||||
|
|
@ -37,7 +37,7 @@ Create or update fallbacks for a specific model.
|
|||
**Response:**
|
||||
```json
|
||||
{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"fallback_models": ["gpt-4", "claude-3-haiku"],
|
||||
"fallback_type": "general",
|
||||
"message": "Fallback configuration created successfully"
|
||||
|
|
@ -50,7 +50,7 @@ curl -X POST "http://localhost:4000/fallback" \
|
|||
-H "Authorization: Bearer sk-1234" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"fallback_models": ["gpt-4", "claude-3-haiku"],
|
||||
"fallback_type": "general"
|
||||
}'
|
||||
|
|
@ -67,7 +67,7 @@ response = requests.post(
|
|||
"Content-Type": "application/json"
|
||||
},
|
||||
json={
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"fallback_models": ["gpt-4", "claude-3-haiku"],
|
||||
"fallback_type": "general"
|
||||
}
|
||||
|
|
@ -87,7 +87,7 @@ Get fallback configuration for a specific model.
|
|||
**Response:**
|
||||
```json
|
||||
{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"fallback_models": ["gpt-4", "claude-3-haiku"],
|
||||
"fallback_type": "general"
|
||||
}
|
||||
|
|
@ -95,7 +95,7 @@ Get fallback configuration for a specific model.
|
|||
|
||||
**Example using cURL:**
|
||||
```bash
|
||||
curl -X GET "http://localhost:4000/fallback/gpt-3.5-turbo?fallback_type=general" \
|
||||
curl -X GET "http://localhost:4000/fallback/gpt-4o?fallback_type=general" \
|
||||
-H "Authorization: Bearer sk-1234"
|
||||
```
|
||||
|
||||
|
|
@ -104,7 +104,7 @@ curl -X GET "http://localhost:4000/fallback/gpt-3.5-turbo?fallback_type=general"
|
|||
import requests
|
||||
|
||||
response = requests.get(
|
||||
"http://localhost:4000/fallback/gpt-3.5-turbo",
|
||||
"http://localhost:4000/fallback/gpt-4o",
|
||||
headers={"Authorization": "Bearer sk-1234"},
|
||||
params={"fallback_type": "general"}
|
||||
)
|
||||
|
|
@ -123,7 +123,7 @@ Delete fallback configuration for a specific model.
|
|||
**Response:**
|
||||
```json
|
||||
{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"fallback_type": "general",
|
||||
"message": "Fallback configuration deleted successfully"
|
||||
}
|
||||
|
|
@ -131,7 +131,7 @@ Delete fallback configuration for a specific model.
|
|||
|
||||
**Example using cURL:**
|
||||
```bash
|
||||
curl -X DELETE "http://localhost:4000/fallback/gpt-3.5-turbo?fallback_type=general" \
|
||||
curl -X DELETE "http://localhost:4000/fallback/gpt-4o?fallback_type=general" \
|
||||
-H "Authorization: Bearer sk-1234"
|
||||
```
|
||||
|
||||
|
|
@ -140,7 +140,7 @@ curl -X DELETE "http://localhost:4000/fallback/gpt-3.5-turbo?fallback_type=gener
|
|||
import requests
|
||||
|
||||
response = requests.delete(
|
||||
"http://localhost:4000/fallback/gpt-3.5-turbo",
|
||||
"http://localhost:4000/fallback/gpt-4o",
|
||||
headers={"Authorization": "Bearer sk-1234"},
|
||||
params={"fallback_type": "general"}
|
||||
)
|
||||
|
|
@ -155,7 +155,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
|||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -186,7 +186,7 @@ The endpoints perform the following validations:
|
|||
{
|
||||
"detail": {
|
||||
"error": "Invalid fallback models: ['non-existent-model']",
|
||||
"available_models": ["gpt-3.5-turbo", "gpt-4", "claude-3-haiku"]
|
||||
"available_models": ["gpt-4o", "gpt-4", "claude-3-haiku"]
|
||||
}
|
||||
}
|
||||
```
|
||||
|
|
@ -195,7 +195,7 @@ The endpoints perform the following validations:
|
|||
```json
|
||||
{
|
||||
"detail": {
|
||||
"error": "Model 'gpt-3.5-turbo' not found in router",
|
||||
"error": "Model 'gpt-4o' not found in router",
|
||||
"available_models": ["gpt-4", "claude-3-haiku"]
|
||||
}
|
||||
}
|
||||
|
|
@ -219,7 +219,7 @@ Used for any type of error that occurs during model invocation. This is the most
|
|||
|
||||
```json
|
||||
{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"fallback_models": ["gpt-4", "claude-3-haiku"],
|
||||
"fallback_type": "general"
|
||||
}
|
||||
|
|
@ -232,7 +232,7 @@ Specifically triggered when a context window exceeded error occurs.
|
|||
|
||||
```json
|
||||
{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"fallback_models": ["gpt-4-32k", "claude-3-opus"],
|
||||
"fallback_type": "context_window"
|
||||
}
|
||||
|
|
|
|||
|
|
@ -37,22 +37,22 @@ Use the `order` parameter to prioritize specific deployments. [See Deployment Or
|
|||
## Quick Start - Load Balancing
|
||||
#### Step 1 - Set deployments on config
|
||||
|
||||
**Example config below**. Here requests with `model=gpt-3.5-turbo` will be routed across multiple instances of `azure/gpt-3.5-turbo`
|
||||
**Example config below**. Here requests with `model=gpt-4o` will be routed across multiple instances of `azure/gpt-4o`
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: azure/<your-deployment-name>
|
||||
api_base: <your-azure-endpoint>
|
||||
api_key: <your-azure-api-key>
|
||||
rpm: 6 # Rate limit for this deployment: in requests per minute (rpm)
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: azure/gpt-turbo-small-ca
|
||||
api_base: https://my-endpoint-canada-berri992.openai.azure.com/
|
||||
api_key: <your-azure-api-key>
|
||||
rpm: 6
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: azure/gpt-turbo-large
|
||||
api_base: https://openai-france-1234.openai.azure.com/
|
||||
|
|
@ -61,7 +61,7 @@ model_list:
|
|||
|
||||
router_settings:
|
||||
routing_strategy: simple-shuffle # Literal["simple-shuffle", "least-busy", "usage-based-routing","latency-based-routing"], default="simple-shuffle"
|
||||
model_group_alias: {"gpt-4": "gpt-3.5-turbo"} # all requests with `gpt-4` will be routed to models with `gpt-3.5-turbo`
|
||||
model_group_alias: {"gpt-4": "gpt-4o"} # all requests with `gpt-4` will be routed to models with `gpt-4o`
|
||||
num_retries: 2
|
||||
timeout: 30 # 30 seconds
|
||||
redis_host: <your redis host> # set this when using multiple litellm proxy deployments, load balancing state stored in redis
|
||||
|
|
@ -142,9 +142,9 @@ $ litellm --config /path/to/config.yaml
|
|||
|
||||
### Test - Simple Call
|
||||
|
||||
Here requests with model=gpt-3.5-turbo will be routed across multiple instances of azure/gpt-3.5-turbo
|
||||
Here requests with model=gpt-4o will be routed across multiple instances of azure/gpt-4o
|
||||
|
||||
👉 Key Change: `model="gpt-3.5-turbo"`
|
||||
👉 Key Change: `model="gpt-4o"`
|
||||
|
||||
**Check the `model_id` in Response Headers to make sure the requests are being load balanced**
|
||||
|
||||
|
|
@ -160,7 +160,7 @@ client = openai.OpenAI(
|
|||
)
|
||||
|
||||
response = client.chat.completions.create(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -179,7 +179,7 @@ print(response)
|
|||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -202,7 +202,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
|||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{"role": "user", "content": "Hi there!"}
|
||||
],
|
||||
|
|
@ -221,13 +221,13 @@ Example config
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: azure/<your-deployment-name>
|
||||
api_base: <your-azure-endpoint>
|
||||
api_key: <your-azure-api-key>
|
||||
rpm: 6 # Rate limit for this deployment: in requests per minute (rpm)
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: azure/gpt-turbo-small-ca
|
||||
api_base: https://my-endpoint-canada-berri992.openai.azure.com/
|
||||
|
|
@ -248,7 +248,7 @@ Expose an 'alias' for a 'model_name' on the proxy server.
|
|||
|
||||
```
|
||||
model_group_alias: {
|
||||
"gpt-4": "gpt-3.5-turbo"
|
||||
"gpt-4": "gpt-4o"
|
||||
}
|
||||
```
|
||||
|
||||
|
|
@ -264,14 +264,14 @@ Example config with `router_settings`
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: azure/<your-deployment-name>
|
||||
api_base: <your-azure-endpoint>
|
||||
api_key: <your-azure-api-key>
|
||||
|
||||
router_settings:
|
||||
model_group_alias: {"gpt-4": "gpt-3.5-turbo"} # all requests with `gpt-4` will be routed to models
|
||||
model_group_alias: {"gpt-4": "gpt-4o"} # all requests with `gpt-4` will be routed to models
|
||||
```
|
||||
|
||||
### Hide Alias Models
|
||||
|
|
@ -284,7 +284,7 @@ Use this if you want to set-up aliases for:
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: azure/<your-deployment-name>
|
||||
api_base: <your-azure-endpoint>
|
||||
|
|
@ -293,7 +293,7 @@ model_list:
|
|||
router_settings:
|
||||
model_group_alias:
|
||||
"GPT-3.5-turbo": # alias
|
||||
model: "gpt-3.5-turbo" # Actual model name in 'model_list'
|
||||
model: "gpt-4o" # Actual model name in 'model_list'
|
||||
hidden: true # Exclude from `/v1/models`, `/v1/model/info`, `/v1/model_group/info`
|
||||
```
|
||||
|
||||
|
|
|
|||
|
|
@ -12,12 +12,12 @@ Set allowed models for a key using the `models` param
|
|||
curl 'http://0.0.0.0:4000/key/generate' \
|
||||
--header 'Authorization: Bearer <your-master-key>' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data-raw '{"models": ["gpt-3.5-turbo", "gpt-4"]}'
|
||||
--data-raw '{"models": ["gpt-4o", "gpt-4"]}'
|
||||
```
|
||||
|
||||
:::info
|
||||
|
||||
This key can only make requests to `models` that are `gpt-3.5-turbo` or `gpt-4`
|
||||
This key can only make requests to `models` that are `gpt-4o` or `gpt-4`
|
||||
|
||||
:::
|
||||
|
||||
|
|
@ -195,7 +195,7 @@ When `include_metadata=true` is specified, the response includes fallback inform
|
|||
"created": 1677610602,
|
||||
"owned_by": "openai",
|
||||
"fallbacks": {
|
||||
"general": ["gpt-3.5-turbo", "claude-3-sonnet"],
|
||||
"general": ["gpt-4o", "claude-3-sonnet"],
|
||||
"context_window": ["gpt-4-turbo", "claude-3-opus"],
|
||||
"content_policy": ["claude-3-haiku"]
|
||||
}
|
||||
|
|
|
|||
|
|
@ -47,7 +47,7 @@ general_settings:
|
|||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
|
|||
|
|
@ -368,7 +368,7 @@ os.environ["LANGFUSE_SECRET_KEY"] = "secret_key" # [OPTIONAL] set here or in `.c
|
|||
litellm.set_verbose = True # see raw request to provider
|
||||
|
||||
resp = litellm.completion(
|
||||
model="langfuse/gpt-3.5-turbo",
|
||||
model="langfuse/gpt-4o",
|
||||
prompt_id="test-chat-prompt",
|
||||
prompt_variables={"user_message": "this is used"}, # [OPTIONAL]
|
||||
messages=[{"role": "user", "content": "<IGNORED>"}],
|
||||
|
|
@ -391,7 +391,7 @@ model_list:
|
|||
api_key: os.environ/OPENAI_API_KEY
|
||||
- model_name: openai-model
|
||||
litellm_params:
|
||||
model: openai/gpt-3.5-turbo
|
||||
model: openai/gpt-4o
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
```
|
||||
|
||||
|
|
@ -435,7 +435,7 @@ client = openai.OpenAI(
|
|||
|
||||
# request sent to model set on litellm proxy, `litellm --model`
|
||||
response = client.chat.completions.create(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -465,7 +465,7 @@ print(response)
|
|||
POST Request Sent from LiteLLM:
|
||||
curl -X POST \
|
||||
https://api.openai.com/v1/ \
|
||||
-d '{'model': 'gpt-3.5-turbo', 'messages': <YOUR LANGFUSE PROMPT TEMPLATE>}'
|
||||
-d '{'model': 'gpt-4o', 'messages': <YOUR LANGFUSE PROMPT TEMPLATE>}'
|
||||
```
|
||||
|
||||
## How to set model
|
||||
|
|
@ -479,7 +479,7 @@ You can do `langfuse/<litellm_model_name>`
|
|||
|
||||
```python
|
||||
litellm.completion(
|
||||
model="langfuse/gpt-3.5-turbo", # or `langfuse/anthropic/claude-3-5-sonnet`
|
||||
model="langfuse/gpt-4o", # or `langfuse/anthropic/claude-3-5-sonnet`
|
||||
...
|
||||
)
|
||||
```
|
||||
|
|
@ -489,9 +489,9 @@ litellm.completion(
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: langfuse/gpt-3.5-turbo # OR langfuse/anthropic/claude-3-5-sonnet
|
||||
model: langfuse/gpt-4o # OR langfuse/anthropic/claude-3-5-sonnet
|
||||
prompt_id: <langfuse_prompt_id>
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
```
|
||||
|
|
@ -507,7 +507,7 @@ If the model is specified in the Langfuse config, it will be used.
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: azure/chatgpt-v-2
|
||||
api_key: os.environ/AZURE_API_KEY
|
||||
|
|
|
|||
|
|
@ -30,7 +30,7 @@ curl -X POST http://localhost:4000/chat/completions \
|
|||
-H "Authorization: Bearer sk-1234" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [{"role": "user", "content": "Hello"}],
|
||||
"metadata": {
|
||||
"tags": ["custom-tag"] # This will be rejected
|
||||
|
|
@ -56,7 +56,7 @@ curl -X POST http://localhost:4000/chat/completions \
|
|||
-H "Authorization: Bearer sk-1234" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [{"role": "user", "content": "Hello"}],
|
||||
"metadata": {
|
||||
"custom_field": "value" # Other metadata fields are allowed
|
||||
|
|
@ -89,9 +89,9 @@ These tags will be automatically inherited by all requests made with that API ke
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
model: gpt-4o
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
general_settings:
|
||||
|
|
|
|||
|
|
@ -361,7 +361,7 @@ litellm_settings:
|
|||
default_team_params: # Default Params to apply when litellm auto creates a team from SSO IDP provider
|
||||
max_budget: 100 # Optional[float], optional): $100 budget for the team
|
||||
budget_duration: 30d # Optional[str], optional): 30 days budget_duration for the team
|
||||
models: ["gpt-3.5-turbo"] # Optional[List[str]], optional): models to be used by the team
|
||||
models: ["gpt-4o"] # Optional[List[str]], optional): models to be used by the team
|
||||
```
|
||||
|
||||
|
||||
|
|
@ -384,7 +384,7 @@ litellm_settings:
|
|||
user_role: "internal_user" # one of "internal_user", "internal_user_viewer", "proxy_admin", "proxy_admin_viewer". New SSO users not in litellm will be created as this user
|
||||
max_budget: 100 # Optional[float], optional): $100 budget for a new SSO sign in user
|
||||
budget_duration: 30d # Optional[str], optional): 30 days budget_duration for a new SSO sign in user
|
||||
models: ["gpt-3.5-turbo"] # Optional[List[str]], optional): models to be used by a new SSO sign in user
|
||||
models: ["gpt-4o"] # Optional[List[str]], optional): models to be used by a new SSO sign in user
|
||||
teams: # Optional[List[NewUserRequestTeam]], optional): teams to be used by the user
|
||||
- team_id: "team_id_1" # Required[str]: team_id to be used by the user
|
||||
max_budget_in_team: 100 # Optional[float], optional): $100 budget for the team. Defaults to None.
|
||||
|
|
@ -393,7 +393,7 @@ litellm_settings:
|
|||
default_team_params: # Default Params to apply when litellm auto creates a team from SSO IDP provider
|
||||
max_budget: 100 # Optional[float], optional): $100 budget for the team
|
||||
budget_duration: 30d # Optional[str], optional): 30 days budget_duration for the team
|
||||
models: ["gpt-3.5-turbo"] # Optional[List[str]], optional): models to be used by the team
|
||||
models: ["gpt-4o"] # Optional[List[str]], optional): models to be used by the team
|
||||
|
||||
|
||||
upperbound_key_generate_params: # Upperbound for /key/generate requests when self-serve flow is on
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue