diff --git a/docs/my-website/docs/proxy/arize_phoenix_prompts.md b/docs/my-website/docs/proxy/arize_phoenix_prompts.md index 138074b1bc3..0a22974372b 100644 --- a/docs/my-website/docs/proxy/arize_phoenix_prompts.md +++ b/docs/my-website/docs/proxy/arize_phoenix_prompts.md @@ -42,7 +42,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \ -H 'Content-Type: application/json' \ -H 'Authorization: Bearer sk-1234' \ -d '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "prompt_id": "simple_prompt", "prompt_variables": { "question": "Explain quantum computing" diff --git a/docs/my-website/docs/proxy/cli.md b/docs/my-website/docs/proxy/cli.md index d3624000a32..eeac807d2e6 100644 --- a/docs/my-website/docs/proxy/cli.md +++ b/docs/my-website/docs/proxy/cli.md @@ -166,7 +166,7 @@ This page documents all command-line interface (CLI) arguments available for the - The model name to pass to LiteLLM. - **Usage:** ```shell - litellm --model gpt-3.5-turbo + litellm --model gpt-4o ``` ### --alias @@ -214,7 +214,7 @@ This page documents all command-line interface (CLI) arguments available for the - Save the model-specific config. - **Usage:** ```shell - litellm --model gpt-3.5-turbo --save + litellm --model gpt-4o --save ``` ## Model Parameters diff --git a/docs/my-website/docs/proxy/custom_auth.md b/docs/my-website/docs/proxy/custom_auth.md index 3d46e1074cc..f30d6bbd3f4 100644 --- a/docs/my-website/docs/proxy/custom_auth.md +++ b/docs/my-website/docs/proxy/custom_auth.md @@ -182,7 +182,7 @@ async def user_api_key_auth(request: Request, api_key: str) -> UserAPIKeyAuth: soft_budget=800.0, tpm_limit=10000, rpm_limit=100, - models=["gpt-4", "claude-3-sonnet", "gpt-3.5-turbo"], + models=["gpt-4", "claude-3-sonnet", "gpt-4o"], allowed_routes=["/chat/completions", "/embeddings"], expires=datetime.now() + timedelta(days=30), metadata={"department": "engineering", "cost_center": "ai_ops"} @@ -198,7 +198,7 @@ async def user_api_key_auth(request: Request, api_key: str) -> UserAPIKeyAuth: max_budget=100.0, tpm_limit=1000, rpm_limit=20, - models=["gpt-3.5-turbo", "claude-3-haiku"], + models=["gpt-4o", "claude-3-haiku"], team_member_tpm_limit=500, # Limit within team end_user_tpm_limit=100, # Per end-user limit metadata={"project": "chatbot_v2"} @@ -218,7 +218,7 @@ e.g. if they're both in the same dir - `./config.yaml` and `./custom_auth.py`, t model_list: - model_name: "openai-model" litellm_params: - model: "gpt-3.5-turbo" + model: "gpt-4o" litellm_settings: drop_params: True @@ -286,7 +286,7 @@ Key change set `mode: auto`. This will check both litellm api key auth + custom model_list: - model_name: "openai-model" litellm_params: - model: "gpt-3.5-turbo" + model: "gpt-4o" api_key: os.environ/OPENAI_API_KEY general_settings: diff --git a/docs/my-website/docs/proxy/customer_routing.md b/docs/my-website/docs/proxy/customer_routing.md index 9bba5e7235f..593a58ec4a7 100644 --- a/docs/my-website/docs/proxy/customer_routing.md +++ b/docs/my-website/docs/proxy/customer_routing.md @@ -37,13 +37,13 @@ Supported regions are 'eu' and 'us'. ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: model: azure/gpt-35-turbo # 👈 EU azure model api_base: https://my-endpoint-europe-berri-992.openai.azure.com/ api_key: os.environ/AZURE_EUROPE_API_KEY region_name: "eu" - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: model: azure/chatgpt-v-2 api_base: https://openai-gpt-4-test-v-1.openai.azure.com/ @@ -70,7 +70,7 @@ curl -X POST --location 'http://localhost:4000/chat/completions' \ --header 'Content-Type: application/json' \ --header 'Authorization: Bearer sk-1234' \ --data '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", diff --git a/docs/my-website/docs/proxy/customers.md b/docs/my-website/docs/proxy/customers.md index 50a5f994fad..ae79242f298 100644 --- a/docs/my-website/docs/proxy/customers.md +++ b/docs/my-website/docs/proxy/customers.md @@ -391,7 +391,7 @@ curl -X POST 'http://localhost:4000/chat/completions' \ -H 'Content-Type: application/json' \ -H 'Authorization: Bearer sk-1234' \ -d '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [{"role": "user", "content": "Hello"}], "user": "my-customer-id" }' @@ -508,7 +508,7 @@ client = OpenAI( ) completion = client.chat.completions.create( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[ {"role": "system", "content": "You are a helpful assistant."}, {"role": "user", "content": "Hello!"} diff --git a/docs/my-website/docs/proxy/dynamic_rate_limit.md b/docs/my-website/docs/proxy/dynamic_rate_limit.md index 09a111f7297..2884b088881 100644 --- a/docs/my-website/docs/proxy/dynamic_rate_limit.md +++ b/docs/my-website/docs/proxy/dynamic_rate_limit.md @@ -15,7 +15,7 @@ Dynamically allocate TPM/RPM quota to api keys, based on active keys in that min model_list: - model_name: my-fake-model litellm_params: - model: gpt-3.5-turbo + model: gpt-4o api_key: my-fake-key mock_response: hello-world tpm: 60 @@ -130,9 +130,9 @@ Priority reservation allocates a percentage of your model's total TPM/RPM to spe ```yaml showLineNumbers title="config.yaml" model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: - model: "gpt-3.5-turbo" + model: "gpt-4o" api_key: os.environ/OPENAI_API_KEY rpm: 10 # Total model capacity @@ -262,7 +262,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \ -H 'Content-Type: application/json' \ -H 'Authorization: Bearer sk-prod-key' \ -d '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [{"role": "user", "content": "Hello from prod"}] }' ``` @@ -273,7 +273,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \ -H 'Content-Type: application/json' \ -H 'Authorization: Bearer sk-dev-key' \ -d '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [{"role": "user", "content": "Hello from dev"}] }' ``` diff --git a/docs/my-website/docs/proxy/endpoint_activity.md b/docs/my-website/docs/proxy/endpoint_activity.md index d06727ce4b4..655ffcfeb66 100644 --- a/docs/my-website/docs/proxy/endpoint_activity.md +++ b/docs/my-website/docs/proxy/endpoint_activity.md @@ -32,7 +32,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \ # 👈 ENDPOINT AUTOMATICA --header 'Content-Type: application/json' \ --header 'Authorization: Bearer sk-1234' \ # 👈 YOUR PROXY KEY --data '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", diff --git a/docs/my-website/docs/proxy/enterprise.md b/docs/my-website/docs/proxy/enterprise.md index 4b525837a20..353d317cc1b 100644 --- a/docs/my-website/docs/proxy/enterprise.md +++ b/docs/my-website/docs/proxy/enterprise.md @@ -103,7 +103,7 @@ curl --location 'http://localhost:4000/chat/completions' \ --header 'Authorization: Bearer sk-5fmYeaUEbAMpwBNT-QpxyA' \ --header 'Content-Type: application/json' \ --data '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", @@ -128,7 +128,7 @@ curl --location 'http://localhost:4000/chat/completions' \ --header 'Authorization: Bearer sk-5fmYeaUEbAMpwBNT-QpxyA' \ --header 'Content-Type: application/json' \ --data '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "user": "gm", "messages": [ { @@ -155,7 +155,7 @@ curl --location 'http://localhost:4000/chat/completions' \ --header 'Authorization: Bearer sk-5fmYeaUEbAMpwBNT-QpxyA' \ --header 'Content-Type: application/json' \ --data '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "user": "gm", "messages": [ { @@ -170,7 +170,7 @@ curl --location 'http://localhost:4000/chat/completions' \ Expected Response ```shell -{"id":"chatcmpl-9XALnHqkCBMBKrOx7Abg0hURHqYtY","choices":[{"finish_reason":"stop","index":0,"message":{"content":"Hello! How can I assist you today?","role":"assistant"}}],"created":1717691639,"model":"gpt-3.5-turbo-0125","object":"chat.completion","system_fingerprint":null,"usage":{"completion_tokens":9,"prompt_tokens":8,"total_tokens":17}}% +{"id":"chatcmpl-9XALnHqkCBMBKrOx7Abg0hURHqYtY","choices":[{"finish_reason":"stop","index":0,"message":{"content":"Hello! How can I assist you today?","role":"assistant"}}],"created":1717691639,"model":"gpt-4o-0125","object":"chat.completion","system_fingerprint":null,"usage":{"completion_tokens":9,"prompt_tokens":8,"total_tokens":17}}% ``` @@ -481,7 +481,7 @@ client = openai.OpenAI( # request sent to model set on litellm proxy, `litellm --model` response = client.chat.completions.create( - model="gpt-3.5-turbo", + model="gpt-4o", messages = [ { "role": "user", @@ -668,7 +668,7 @@ client = openai.OpenAI( # request sent to model set on litellm proxy, `litellm --model` response = client.chat.completions.create( - model="gpt-3.5-turbo", + model="gpt-4o", messages = [ { "role": "user", @@ -688,7 +688,7 @@ print(response) curl --location 'http://0.0.0.0:4000/chat/completions' \ --header 'Content-Type: application/json' \ --data ' { - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", @@ -748,7 +748,7 @@ litellm_settings: curl --location 'http://0.0.0.0:4000/chat/completions' \ --header 'Content-Type: application/json' \ --data ' { - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", diff --git a/docs/my-website/docs/proxy/fallback_management.md b/docs/my-website/docs/proxy/fallback_management.md index 9e565fee133..6766fa06479 100644 --- a/docs/my-website/docs/proxy/fallback_management.md +++ b/docs/my-website/docs/proxy/fallback_management.md @@ -20,7 +20,7 @@ Create or update fallbacks for a specific model. **Request Body:** ```json { - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "fallback_models": ["gpt-4", "claude-3-haiku"], "fallback_type": "general" } @@ -37,7 +37,7 @@ Create or update fallbacks for a specific model. **Response:** ```json { - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "fallback_models": ["gpt-4", "claude-3-haiku"], "fallback_type": "general", "message": "Fallback configuration created successfully" @@ -50,7 +50,7 @@ curl -X POST "http://localhost:4000/fallback" \ -H "Authorization: Bearer sk-1234" \ -H "Content-Type: application/json" \ -d '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "fallback_models": ["gpt-4", "claude-3-haiku"], "fallback_type": "general" }' @@ -67,7 +67,7 @@ response = requests.post( "Content-Type": "application/json" }, json={ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "fallback_models": ["gpt-4", "claude-3-haiku"], "fallback_type": "general" } @@ -87,7 +87,7 @@ Get fallback configuration for a specific model. **Response:** ```json { - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "fallback_models": ["gpt-4", "claude-3-haiku"], "fallback_type": "general" } @@ -95,7 +95,7 @@ Get fallback configuration for a specific model. **Example using cURL:** ```bash -curl -X GET "http://localhost:4000/fallback/gpt-3.5-turbo?fallback_type=general" \ +curl -X GET "http://localhost:4000/fallback/gpt-4o?fallback_type=general" \ -H "Authorization: Bearer sk-1234" ``` @@ -104,7 +104,7 @@ curl -X GET "http://localhost:4000/fallback/gpt-3.5-turbo?fallback_type=general" import requests response = requests.get( - "http://localhost:4000/fallback/gpt-3.5-turbo", + "http://localhost:4000/fallback/gpt-4o", headers={"Authorization": "Bearer sk-1234"}, params={"fallback_type": "general"} ) @@ -123,7 +123,7 @@ Delete fallback configuration for a specific model. **Response:** ```json { - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "fallback_type": "general", "message": "Fallback configuration deleted successfully" } @@ -131,7 +131,7 @@ Delete fallback configuration for a specific model. **Example using cURL:** ```bash -curl -X DELETE "http://localhost:4000/fallback/gpt-3.5-turbo?fallback_type=general" \ +curl -X DELETE "http://localhost:4000/fallback/gpt-4o?fallback_type=general" \ -H "Authorization: Bearer sk-1234" ``` @@ -140,7 +140,7 @@ curl -X DELETE "http://localhost:4000/fallback/gpt-3.5-turbo?fallback_type=gener import requests response = requests.delete( - "http://localhost:4000/fallback/gpt-3.5-turbo", + "http://localhost:4000/fallback/gpt-4o", headers={"Authorization": "Bearer sk-1234"}, params={"fallback_type": "general"} ) @@ -155,7 +155,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \ -H 'Content-Type: application/json' \ -H 'Authorization: Bearer sk-1234' \ -d '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", @@ -186,7 +186,7 @@ The endpoints perform the following validations: { "detail": { "error": "Invalid fallback models: ['non-existent-model']", - "available_models": ["gpt-3.5-turbo", "gpt-4", "claude-3-haiku"] + "available_models": ["gpt-4o", "gpt-4", "claude-3-haiku"] } } ``` @@ -195,7 +195,7 @@ The endpoints perform the following validations: ```json { "detail": { - "error": "Model 'gpt-3.5-turbo' not found in router", + "error": "Model 'gpt-4o' not found in router", "available_models": ["gpt-4", "claude-3-haiku"] } } @@ -219,7 +219,7 @@ Used for any type of error that occurs during model invocation. This is the most ```json { - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "fallback_models": ["gpt-4", "claude-3-haiku"], "fallback_type": "general" } @@ -232,7 +232,7 @@ Specifically triggered when a context window exceeded error occurs. ```json { - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "fallback_models": ["gpt-4-32k", "claude-3-opus"], "fallback_type": "context_window" } diff --git a/docs/my-website/docs/proxy/load_balancing.md b/docs/my-website/docs/proxy/load_balancing.md index 5bf39d179f6..a38dfb94616 100644 --- a/docs/my-website/docs/proxy/load_balancing.md +++ b/docs/my-website/docs/proxy/load_balancing.md @@ -37,22 +37,22 @@ Use the `order` parameter to prioritize specific deployments. [See Deployment Or ## Quick Start - Load Balancing #### Step 1 - Set deployments on config -**Example config below**. Here requests with `model=gpt-3.5-turbo` will be routed across multiple instances of `azure/gpt-3.5-turbo` +**Example config below**. Here requests with `model=gpt-4o` will be routed across multiple instances of `azure/gpt-4o` ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: model: azure/ api_base: api_key: rpm: 6 # Rate limit for this deployment: in requests per minute (rpm) - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: model: azure/gpt-turbo-small-ca api_base: https://my-endpoint-canada-berri992.openai.azure.com/ api_key: rpm: 6 - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: model: azure/gpt-turbo-large api_base: https://openai-france-1234.openai.azure.com/ @@ -61,7 +61,7 @@ model_list: router_settings: routing_strategy: simple-shuffle # Literal["simple-shuffle", "least-busy", "usage-based-routing","latency-based-routing"], default="simple-shuffle" - model_group_alias: {"gpt-4": "gpt-3.5-turbo"} # all requests with `gpt-4` will be routed to models with `gpt-3.5-turbo` + model_group_alias: {"gpt-4": "gpt-4o"} # all requests with `gpt-4` will be routed to models with `gpt-4o` num_retries: 2 timeout: 30 # 30 seconds redis_host: # set this when using multiple litellm proxy deployments, load balancing state stored in redis @@ -142,9 +142,9 @@ $ litellm --config /path/to/config.yaml ### Test - Simple Call -Here requests with model=gpt-3.5-turbo will be routed across multiple instances of azure/gpt-3.5-turbo +Here requests with model=gpt-4o will be routed across multiple instances of azure/gpt-4o -👉 Key Change: `model="gpt-3.5-turbo"` +👉 Key Change: `model="gpt-4o"` **Check the `model_id` in Response Headers to make sure the requests are being load balanced** @@ -160,7 +160,7 @@ client = openai.OpenAI( ) response = client.chat.completions.create( - model="gpt-3.5-turbo", + model="gpt-4o", messages = [ { "role": "user", @@ -179,7 +179,7 @@ print(response) curl --location 'http://0.0.0.0:4000/chat/completions' \ --header 'Content-Type: application/json' \ --data '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", @@ -202,7 +202,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \ -H 'Content-Type: application/json' \ -H 'Authorization: Bearer sk-1234' \ -d '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ {"role": "user", "content": "Hi there!"} ], @@ -221,13 +221,13 @@ Example config ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: model: azure/ api_base: api_key: rpm: 6 # Rate limit for this deployment: in requests per minute (rpm) - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: model: azure/gpt-turbo-small-ca api_base: https://my-endpoint-canada-berri992.openai.azure.com/ @@ -248,7 +248,7 @@ Expose an 'alias' for a 'model_name' on the proxy server. ``` model_group_alias: { - "gpt-4": "gpt-3.5-turbo" + "gpt-4": "gpt-4o" } ``` @@ -264,14 +264,14 @@ Example config with `router_settings` ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: model: azure/ api_base: api_key: router_settings: - model_group_alias: {"gpt-4": "gpt-3.5-turbo"} # all requests with `gpt-4` will be routed to models + model_group_alias: {"gpt-4": "gpt-4o"} # all requests with `gpt-4` will be routed to models ``` ### Hide Alias Models @@ -284,7 +284,7 @@ Use this if you want to set-up aliases for: ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: model: azure/ api_base: @@ -293,7 +293,7 @@ model_list: router_settings: model_group_alias: "GPT-3.5-turbo": # alias - model: "gpt-3.5-turbo" # Actual model name in 'model_list' + model: "gpt-4o" # Actual model name in 'model_list' hidden: true # Exclude from `/v1/models`, `/v1/model/info`, `/v1/model_group/info` ``` diff --git a/docs/my-website/docs/proxy/model_access.md b/docs/my-website/docs/proxy/model_access.md index 961207cad5a..e65f9251a66 100644 --- a/docs/my-website/docs/proxy/model_access.md +++ b/docs/my-website/docs/proxy/model_access.md @@ -12,12 +12,12 @@ Set allowed models for a key using the `models` param curl 'http://0.0.0.0:4000/key/generate' \ --header 'Authorization: Bearer ' \ --header 'Content-Type: application/json' \ ---data-raw '{"models": ["gpt-3.5-turbo", "gpt-4"]}' +--data-raw '{"models": ["gpt-4o", "gpt-4"]}' ``` :::info -This key can only make requests to `models` that are `gpt-3.5-turbo` or `gpt-4` +This key can only make requests to `models` that are `gpt-4o` or `gpt-4` ::: @@ -195,7 +195,7 @@ When `include_metadata=true` is specified, the response includes fallback inform "created": 1677610602, "owned_by": "openai", "fallbacks": { - "general": ["gpt-3.5-turbo", "claude-3-sonnet"], + "general": ["gpt-4o", "claude-3-sonnet"], "context_window": ["gpt-4-turbo", "claude-3-opus"], "content_policy": ["claude-3-haiku"] } diff --git a/docs/my-website/docs/proxy/oauth2.md b/docs/my-website/docs/proxy/oauth2.md index 41c4110e447..bcab2bd3440 100644 --- a/docs/my-website/docs/proxy/oauth2.md +++ b/docs/my-website/docs/proxy/oauth2.md @@ -47,7 +47,7 @@ general_settings: curl --location 'http://0.0.0.0:4000/chat/completions' \ --header 'Content-Type: application/json' \ --data '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", diff --git a/docs/my-website/docs/proxy/prompt_management.md b/docs/my-website/docs/proxy/prompt_management.md index 08307ba99ec..5dc63462223 100644 --- a/docs/my-website/docs/proxy/prompt_management.md +++ b/docs/my-website/docs/proxy/prompt_management.md @@ -368,7 +368,7 @@ os.environ["LANGFUSE_SECRET_KEY"] = "secret_key" # [OPTIONAL] set here or in `.c litellm.set_verbose = True # see raw request to provider resp = litellm.completion( - model="langfuse/gpt-3.5-turbo", + model="langfuse/gpt-4o", prompt_id="test-chat-prompt", prompt_variables={"user_message": "this is used"}, # [OPTIONAL] messages=[{"role": "user", "content": ""}], @@ -391,7 +391,7 @@ model_list: api_key: os.environ/OPENAI_API_KEY - model_name: openai-model litellm_params: - model: openai/gpt-3.5-turbo + model: openai/gpt-4o api_key: os.environ/OPENAI_API_KEY ``` @@ -435,7 +435,7 @@ client = openai.OpenAI( # request sent to model set on litellm proxy, `litellm --model` response = client.chat.completions.create( - model="gpt-3.5-turbo", + model="gpt-4o", messages = [ { "role": "user", @@ -465,7 +465,7 @@ print(response) POST Request Sent from LiteLLM: curl -X POST \ https://api.openai.com/v1/ \ --d '{'model': 'gpt-3.5-turbo', 'messages': }' +-d '{'model': 'gpt-4o', 'messages': }' ``` ## How to set model @@ -479,7 +479,7 @@ You can do `langfuse/` ```python litellm.completion( - model="langfuse/gpt-3.5-turbo", # or `langfuse/anthropic/claude-3-5-sonnet` + model="langfuse/gpt-4o", # or `langfuse/anthropic/claude-3-5-sonnet` ... ) ``` @@ -489,9 +489,9 @@ litellm.completion( ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: - model: langfuse/gpt-3.5-turbo # OR langfuse/anthropic/claude-3-5-sonnet + model: langfuse/gpt-4o # OR langfuse/anthropic/claude-3-5-sonnet prompt_id: api_key: os.environ/OPENAI_API_KEY ``` @@ -507,7 +507,7 @@ If the model is specified in the Langfuse config, it will be used. ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: model: azure/chatgpt-v-2 api_key: os.environ/AZURE_API_KEY diff --git a/docs/my-website/docs/proxy/reject_clientside_metadata_tags.md b/docs/my-website/docs/proxy/reject_clientside_metadata_tags.md index 534c65939eb..31072e253c7 100644 --- a/docs/my-website/docs/proxy/reject_clientside_metadata_tags.md +++ b/docs/my-website/docs/proxy/reject_clientside_metadata_tags.md @@ -30,7 +30,7 @@ curl -X POST http://localhost:4000/chat/completions \ -H "Authorization: Bearer sk-1234" \ -H "Content-Type: application/json" \ -d '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [{"role": "user", "content": "Hello"}], "metadata": { "tags": ["custom-tag"] # This will be rejected @@ -56,7 +56,7 @@ curl -X POST http://localhost:4000/chat/completions \ -H "Authorization: Bearer sk-1234" \ -H "Content-Type: application/json" \ -d '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [{"role": "user", "content": "Hello"}], "metadata": { "custom_field": "value" # Other metadata fields are allowed @@ -89,9 +89,9 @@ These tags will be automatically inherited by all requests made with that API ke ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: - model: gpt-3.5-turbo + model: gpt-4o api_key: os.environ/OPENAI_API_KEY general_settings: diff --git a/docs/my-website/docs/proxy/self_serve.md b/docs/my-website/docs/proxy/self_serve.md index b54344c1d05..f723cd3834a 100644 --- a/docs/my-website/docs/proxy/self_serve.md +++ b/docs/my-website/docs/proxy/self_serve.md @@ -361,7 +361,7 @@ litellm_settings: default_team_params: # Default Params to apply when litellm auto creates a team from SSO IDP provider max_budget: 100 # Optional[float], optional): $100 budget for the team budget_duration: 30d # Optional[str], optional): 30 days budget_duration for the team - models: ["gpt-3.5-turbo"] # Optional[List[str]], optional): models to be used by the team + models: ["gpt-4o"] # Optional[List[str]], optional): models to be used by the team ``` @@ -384,7 +384,7 @@ litellm_settings: user_role: "internal_user" # one of "internal_user", "internal_user_viewer", "proxy_admin", "proxy_admin_viewer". New SSO users not in litellm will be created as this user max_budget: 100 # Optional[float], optional): $100 budget for a new SSO sign in user budget_duration: 30d # Optional[str], optional): 30 days budget_duration for a new SSO sign in user - models: ["gpt-3.5-turbo"] # Optional[List[str]], optional): models to be used by a new SSO sign in user + models: ["gpt-4o"] # Optional[List[str]], optional): models to be used by a new SSO sign in user teams: # Optional[List[NewUserRequestTeam]], optional): teams to be used by the user - team_id: "team_id_1" # Required[str]: team_id to be used by the user max_budget_in_team: 100 # Optional[float], optional): $100 budget for the team. Defaults to None. @@ -393,7 +393,7 @@ litellm_settings: default_team_params: # Default Params to apply when litellm auto creates a team from SSO IDP provider max_budget: 100 # Optional[float], optional): $100 budget for the team budget_duration: 30d # Optional[str], optional): 30 days budget_duration for the team - models: ["gpt-3.5-turbo"] # Optional[List[str]], optional): models to be used by the team + models: ["gpt-4o"] # Optional[List[str]], optional): models to be used by the team upperbound_key_generate_params: # Upperbound for /key/generate requests when self-serve flow is on