From bb741e73fcaeb694715f073d1ad44dc9721c0bcd Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Sat, 21 Mar 2026 17:59:29 +0000 Subject: [PATCH] docs(proxy): replace gpt-3.5-turbo with gpt-4o in proxy guides Update example model names across virtual keys, users, keys, onboarding, reliability, projects, team routing, model management, logging, timeouts, provider budgets, service accounts, pass-through, tag routing, and token auth. Co-authored-by: Krish Dholakia --- docs/my-website/docs/proxy/logging.md | 100 +++++++++--------- .../my-website/docs/proxy/model_management.md | 4 +- docs/my-website/docs/proxy/pass_through.md | 2 +- .../docs/proxy/project_management.md | 16 +-- .../docs/proxy/provider_budget_routing.md | 8 +- docs/my-website/docs/proxy/reliability.md | 80 +++++++------- .../my-website/docs/proxy/service_accounts.md | 6 +- docs/my-website/docs/proxy/tag_routing.md | 2 +- .../docs/proxy/team_based_routing.md | 8 +- docs/my-website/docs/proxy/timeout.md | 14 +-- docs/my-website/docs/proxy/token_auth.md | 8 +- docs/my-website/docs/proxy/user_keys.md | 32 +++--- docs/my-website/docs/proxy/user_onboarding.md | 2 +- docs/my-website/docs/proxy/users.md | 20 ++-- docs/my-website/docs/proxy/virtual_keys.md | 22 ++-- 15 files changed, 162 insertions(+), 162 deletions(-) diff --git a/docs/my-website/docs/proxy/logging.md b/docs/my-website/docs/proxy/logging.md index 74a79776fbd..3b07dc718ac 100644 --- a/docs/my-website/docs/proxy/logging.md +++ b/docs/my-website/docs/proxy/logging.md @@ -34,7 +34,7 @@ curl -i -sSL --location 'http://0.0.0.0:4000/chat/completions' \ --header 'Authorization: Bearer sk-1234' \ --header 'Content-Type: application/json' \ --data '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [{"role": "user", "content": "what llm are you"}] }' | grep 'x-litellm' ``` @@ -70,9 +70,9 @@ Set `litellm.turn_off_message_logging=True` This will prevent the messages and r **1. Setup config.yaml** ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: - model: gpt-3.5-turbo + model: gpt-4o litellm_settings: success_callback: ["langfuse"] turn_off_message_logging: True # 👈 Key Change @@ -83,7 +83,7 @@ litellm_settings: curl --location 'http://0.0.0.0:4000/chat/completions' \ --header 'Content-Type: application/json' \ --data '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", @@ -116,9 +116,9 @@ Example config.yaml ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: - model: gpt-3.5-turbo + model: gpt-4o ``` **2. Setup per request header** @@ -129,7 +129,7 @@ curl -L -X POST 'http://0.0.0.0:4000/v1/chat/completions' \ -H 'Authorization: Bearer sk-zV5HlSIm8ihj1F9C_ZbB1g' \ -H 'x-litellm-enable-message-redaction: true' \ -d '{ - "model": "gpt-3.5-turbo-testing", + "model": "gpt-4o-testing", "messages": [ { "role": "user", @@ -176,7 +176,7 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \ --header 'Content-Type: application/json' \ --header 'LiteLLM-Disable-Message-Redaction: true' \ --data '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", @@ -209,7 +209,7 @@ curl -L -X POST 'http://0.0.0.0:4000/v1/chat/completions' \ -H 'Content-Type: application/json' \ -H 'Authorization: Bearer ' \ -d '{ - "model": "openai/gpt-3.5-turbo", + "model": "openai/gpt-4o", "messages": [ { "role": "user", @@ -238,7 +238,7 @@ client = openai.OpenAI( # request sent to model set on litellm proxy, `litellm --model` response = client.chat.completions.create( - model="gpt-3.5-turbo", + model="gpt-4o", messages = [ { "role": "user", @@ -358,9 +358,9 @@ pip install langfuse>=2.0.0 ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: - model: gpt-3.5-turbo + model: gpt-4o litellm_settings: success_callback: ["langfuse"] ``` @@ -404,7 +404,7 @@ Pass `metadata` as part of the request body curl --location 'http://0.0.0.0:4000/chat/completions' \ --header 'Content-Type: application/json' \ --data '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", @@ -434,7 +434,7 @@ client = openai.OpenAI( # request sent to model set on litellm proxy, `litellm --model` response = client.chat.completions.create( - model="gpt-3.5-turbo", + model="gpt-4o", messages = [ { "role": "user", @@ -468,7 +468,7 @@ from langchain.schema import HumanMessage, SystemMessage chat = ChatOpenAI( openai_api_base="http://0.0.0.0:4000", - model = "gpt-3.5-turbo", + model = "gpt-4o", temperature=0.1, extra_body={ "metadata": { @@ -648,7 +648,7 @@ Pass `metadata` as part of the request body curl --location 'http://0.0.0.0:4000/chat/completions' \ --header 'Content-Type: application/json' \ --data '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", @@ -675,7 +675,7 @@ client = openai.OpenAI( # request sent to model set on litellm proxy, `litellm --model` response = client.chat.completions.create( - model="gpt-3.5-turbo", + model="gpt-4o", messages = [ { "role": "user", @@ -706,7 +706,7 @@ from langchain.schema import HumanMessage, SystemMessage chat = ChatOpenAI( openai_api_base="http://0.0.0.0:4000", - model = "gpt-3.5-turbo", + model = "gpt-4o", temperature=0.1, extra_body={ "metadata": { @@ -783,7 +783,7 @@ Test Request curl --location 'http://0.0.0.0:4000/chat/completions' \ --header 'Content-Type: application/json' \ --data ' { - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", @@ -863,7 +863,7 @@ Test Request curl --location 'http://0.0.0.0:4000/chat/completions' \ --header 'Content-Type: application/json' \ --data ' { - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", @@ -909,7 +909,7 @@ Test Request curl --location 'http://0.0.0.0:4000/chat/completions' \ --header 'Content-Type: application/json' \ --data ' { - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", @@ -956,7 +956,7 @@ Test Request curl --location 'http://0.0.0.0:4000/chat/completions' \ --header 'Content-Type: application/json' \ --data ' { - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", @@ -1005,7 +1005,7 @@ Test Request curl --location 'http://0.0.0.0:4000/chat/completions' \ --header 'Content-Type: application/json' \ --data ' { - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", @@ -1288,7 +1288,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \ -H 'Content-Type: application/json' \ -H 'Authorization: Bearer sk-1234' \ -d '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "system", @@ -1326,9 +1326,9 @@ AWS_REGION_NAME = "" ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: - model: gpt-3.5-turbo + model: gpt-4o litellm_settings: success_callback: ["s3_v2"] s3_callback_params: @@ -1755,9 +1755,9 @@ In the config below, we pass ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: - model: gpt-3.5-turbo + model: gpt-4o litellm_settings: callbacks: custom_callbacks.proxy_handler_instance # sets litellm.callbacks = [proxy_handler_instance] @@ -1798,9 +1798,9 @@ custom_handler = MyCustomHandler() ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: - model: gpt-3.5-turbo + model: gpt-4o litellm_settings: callbacks: ["s3://litellm-proxy/custom_callbacks.custom_handler"] @@ -1810,9 +1810,9 @@ litellm_settings: ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: - model: gpt-3.5-turbo + model: gpt-4o litellm_settings: callbacks: ["gcs://my-gcs-bucket/custom_callbacks.custom_handler"] @@ -1898,7 +1898,7 @@ litellm --config proxy_config.yaml curl --location 'http://0.0.0.0:4000/chat/completions' \ --header 'Authorization: Bearer sk-1234' \ --data ' { - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", @@ -1914,13 +1914,13 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \ ```shell On Success - Model: gpt-3.5-turbo, + Model: gpt-4o, Messages: [{'role': 'user', 'content': 'good morning good sir'}], User: ishaan-app, Usage: {'completion_tokens': 10, 'prompt_tokens': 11, 'total_tokens': 21}, Cost: 3.65e-05, - Response: {'id': 'chatcmpl-8S8avKJ1aVBg941y5xzGMSKrYCMvN', 'choices': [{'finish_reason': 'stop', 'index': 0, 'message': {'content': 'Good morning! How can I assist you today?', 'role': 'assistant'}}], 'created': 1701716913, 'model': 'gpt-3.5-turbo-0613', 'object': 'chat.completion', 'system_fingerprint': None, 'usage': {'completion_tokens': 10, 'prompt_tokens': 11, 'total_tokens': 21}} - Proxy Metadata: {'user_api_key': None, 'headers': Headers({'host': '0.0.0.0:4000', 'user-agent': 'curl/7.88.1', 'accept': '*/*', 'authorization': 'Bearer sk-1234', 'content-length': '199', 'content-type': 'application/x-www-form-urlencoded'}), 'model_group': 'gpt-3.5-turbo', 'deployment': 'gpt-3.5-turbo-ModelID-gpt-3.5-turbo'} + Response: {'id': 'chatcmpl-8S8avKJ1aVBg941y5xzGMSKrYCMvN', 'choices': [{'finish_reason': 'stop', 'index': 0, 'message': {'content': 'Good morning! How can I assist you today?', 'role': 'assistant'}}], 'created': 1701716913, 'model': 'gpt-4o-0613', 'object': 'chat.completion', 'system_fingerprint': None, 'usage': {'completion_tokens': 10, 'prompt_tokens': 11, 'total_tokens': 21}} + Proxy Metadata: {'user_api_key': None, 'headers': Headers({'host': '0.0.0.0:4000', 'user-agent': 'curl/7.88.1', 'accept': '*/*', 'authorization': 'Bearer sk-1234', 'content-length': '199', 'content-type': 'application/x-www-form-urlencoded'}), 'model_group': 'gpt-4o', 'deployment': 'gpt-4o-ModelID-gpt-4o'} ``` #### Logging Proxy Request Object, Header, Url @@ -2404,9 +2404,9 @@ AWS_REGION_NAME = "" ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: - model: gpt-3.5-turbo + model: gpt-4o litellm_settings: success_callback: ["dynamodb"] dynamodb_table_name: your-table-name @@ -2458,13 +2458,13 @@ Your logs should be available on DynamoDB "S": "{}" }, "model": { - "S": "gpt-3.5-turbo" + "S": "gpt-4o" }, "modelParameters": { "S": "{'temperature': 0.7, 'max_tokens': 100, 'user': 'ishaan-2'}" }, "response": { - "S": "ModelResponse(id='chatcmpl-8W15J4480a3fAQ1yQaMgtsKJAicen', choices=[Choices(finish_reason='stop', index=0, message=Message(content='Great! What can I assist you with?', role='assistant'))], created=1702641357, model='gpt-3.5-turbo-0613', object='chat.completion', system_fingerprint=None, usage=Usage(completion_tokens=9, prompt_tokens=11, total_tokens=20))" + "S": "ModelResponse(id='chatcmpl-8W15J4480a3fAQ1yQaMgtsKJAicen', choices=[Choices(finish_reason='stop', index=0, message=Message(content='Great! What can I assist you with?', role='assistant'))], created=1702641357, model='gpt-4o-0613', object='chat.completion', system_fingerprint=None, usage=Usage(completion_tokens=9, prompt_tokens=11, total_tokens=20))" }, "startTime": { "S": "2023-12-15 17:25:56.047035" @@ -2531,9 +2531,9 @@ export SENTRY_ENVIRONMENT="development" # Controls the Sentry Environment (defau ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: - model: gpt-3.5-turbo + model: gpt-4o litellm_settings: # other settings failure_callback: ["sentry"] @@ -2571,9 +2571,9 @@ ATHINA_API_KEY = "your-athina-api-key" ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: - model: gpt-3.5-turbo + model: gpt-4o litellm_settings: success_callback: ["athina"] ``` @@ -2592,7 +2592,7 @@ Test Request curl --location 'http://0.0.0.0:4000/chat/completions' \ --header 'Content-Type: application/json' \ --data ' { - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", @@ -2625,9 +2625,9 @@ AZURE_CONTENT_SAFETY_KEY = "" ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: - model: gpt-3.5-turbo + model: gpt-4o litellm_settings: callbacks: ["azure_content_safety"] azure_content_safety_params: @@ -2649,7 +2649,7 @@ Test Request curl --location 'http://0.0.0.0:4000/chat/completions' \ --header 'Content-Type: application/json' \ --data ' { - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", @@ -2672,9 +2672,9 @@ You can customize the thresholds for each category by setting the `thresholds` i ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: - model: gpt-3.5-turbo + model: gpt-4o litellm_settings: callbacks: ["azure_content_safety"] azure_content_safety_params: diff --git a/docs/my-website/docs/proxy/model_management.md b/docs/my-website/docs/proxy/model_management.md index 1faaf697d36..09894c822d4 100644 --- a/docs/my-website/docs/proxy/model_management.md +++ b/docs/my-website/docs/proxy/model_management.md @@ -48,14 +48,14 @@ Add a new model to the proxy via the `/model/new` API, to add models without res curl -X POST "http://0.0.0.0:4000/model/new" \ -H "accept: application/json" \ -H "Content-Type: application/json" \ - -d '{ "model_name": "azure-gpt-turbo", "litellm_params": {"model": "azure/gpt-3.5-turbo", "api_key": "os.environ/AZURE_API_KEY", "api_base": "my-azure-api-base"} }' + -d '{ "model_name": "azure-gpt-turbo", "litellm_params": {"model": "azure/gpt-4o", "api_key": "os.environ/AZURE_API_KEY", "api_base": "my-azure-api-base"} }' ``` ```yaml model_list: - - model_name: gpt-3.5-turbo ### RECEIVED MODEL NAME ### `openai.chat.completions.create(model="gpt-3.5-turbo",...)` + - model_name: gpt-4o ### RECEIVED MODEL NAME ### `openai.chat.completions.create(model="gpt-4o",...)` litellm_params: # all params accepted by litellm.completion() - https://github.com/BerriAI/litellm/blob/9b46ec05b02d36d6e4fb5c32321e51e7f56e4a6e/litellm/types/router.py#L297 model: azure/gpt-turbo-small-eu ### MODEL NAME sent to `litellm.completion()` ### api_base: https://my-endpoint-europe-berri-992.openai.azure.com/ diff --git a/docs/my-website/docs/proxy/pass_through.md b/docs/my-website/docs/proxy/pass_through.md index f47d7064140..de7e5d77d3a 100644 --- a/docs/my-website/docs/proxy/pass_through.md +++ b/docs/my-website/docs/proxy/pass_through.md @@ -344,7 +344,7 @@ anthropic_adapter = AnthropicAdapter() model_list: - model_name: my-claude-endpoint litellm_params: - model: gpt-3.5-turbo + model: gpt-4o api_key: os.environ/OPENAI_API_KEY general_settings: diff --git a/docs/my-website/docs/proxy/project_management.md b/docs/my-website/docs/proxy/project_management.md index 06ed5b4a0d5..093887df6d6 100644 --- a/docs/my-website/docs/proxy/project_management.md +++ b/docs/my-website/docs/proxy/project_management.md @@ -41,7 +41,7 @@ curl --location 'http://0.0.0.0:4000/project/new' \ --data '{ "project_alias": "flight-search-assistant", "team_id": "ad898803-c8a3-4f4a-976a-a3c372cffa45", - "models": ["gpt-4", "gpt-3.5-turbo"], + "models": ["gpt-4", "gpt-4o"], "max_budget": 100, "metadata": { "use_case_id": "SNOW-12345", @@ -56,7 +56,7 @@ curl --location 'http://0.0.0.0:4000/project/new' \ "project_id": "e402a141-725a-4437-bff5-d47459189716", "project_alias": "flight-search-assistant", "team_id": "ad898803-c8a3-4f4a-976a-a3c372cffa45", - "models": ["gpt-4", "gpt-3.5-turbo"], + "models": ["gpt-4", "gpt-4o"], "max_budget": 100, ... } @@ -69,7 +69,7 @@ curl 'http://0.0.0.0:4000/key/generate' \ --header 'Authorization: Bearer sk-1234' \ --header 'Content-Type: application/json' \ --data-raw '{ - "models": ["gpt-3.5-turbo", "gpt-4"], + "models": ["gpt-4o", "gpt-4"], "metadata": {"user": "ishaan@berri.ai"}, "project_id": "e402a141-725a-4437-bff5-d47459189716" }' | jq @@ -219,11 +219,11 @@ curl --location 'http://0.0.0.0:4000/project/info?project_id=project-abc' \ "project_id": "project-abc", "project_alias": "flight-search-assistant", "team_id": "team-123", - "models": ["gpt-4", "gpt-3.5-turbo"], + "models": ["gpt-4", "gpt-4o"], "spend": 45.67, "model_spend": { "gpt-4": 42.30, - "gpt-3.5-turbo": 3.37 + "gpt-4o": 3.37 }, "litellm_budget_table": { "budget_id": "budget-xyz", @@ -300,17 +300,17 @@ curl --location 'http://0.0.0.0:4000/project/new' \ --data '{ "project_alias": "multi-model-project", "team_id": "team-123", - "models": ["gpt-4", "gpt-3.5-turbo", "claude-3-sonnet"], + "models": ["gpt-4", "gpt-4o", "claude-3-sonnet"], "max_budget": 500, "metadata": { "model_tpm_limit": { "gpt-4": 50000, - "gpt-3.5-turbo": 200000, + "gpt-4o": 200000, "claude-3-sonnet": 100000 }, "model_rpm_limit": { "gpt-4": 50, - "gpt-3.5-turbo": 500, + "gpt-4o": 500, "claude-3-sonnet": 100 } } diff --git a/docs/my-website/docs/proxy/provider_budget_routing.md b/docs/my-website/docs/proxy/provider_budget_routing.md index ff43d2787a4..820521301ca 100644 --- a/docs/my-website/docs/proxy/provider_budget_routing.md +++ b/docs/my-website/docs/proxy/provider_budget_routing.md @@ -17,9 +17,9 @@ Set provider budgets in your `proxy_config.yaml` file #### Proxy Config setup ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: - model: openai/gpt-3.5-turbo + model: openai/gpt-4o api_key: os.environ/OPENAI_API_KEY router_settings: @@ -366,9 +366,9 @@ If you are using a multi-instance setup, you will need to set the Redis host, po ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: - model: openai/gpt-3.5-turbo + model: openai/gpt-4o api_key: os.environ/OPENAI_API_KEY router_settings: diff --git a/docs/my-website/docs/proxy/reliability.md b/docs/my-website/docs/proxy/reliability.md index d58572cb642..19e9132e08b 100644 --- a/docs/my-website/docs/proxy/reliability.md +++ b/docs/my-website/docs/proxy/reliability.md @@ -19,7 +19,7 @@ Fallbacks are typically done from one `model_name` to another `model_name`. Key change: ```python -fallbacks=[{"gpt-3.5-turbo": ["gpt-4"]}] +fallbacks=[{"gpt-4o": ["gpt-4"]}] ``` @@ -30,7 +30,7 @@ from litellm import Router router = Router( model_list=[ { - "model_name": "gpt-3.5-turbo", + "model_name": "gpt-4o", "litellm_params": { "model": "azure/", "api_base": "", @@ -48,7 +48,7 @@ router = Router( } } ], - fallbacks=[{"gpt-3.5-turbo": ["gpt-4"]}] # 👈 KEY CHANGE + fallbacks=[{"gpt-4o": ["gpt-4"]}] # 👈 KEY CHANGE ) ``` @@ -59,7 +59,7 @@ router = Router( ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: model: azure/ api_base: @@ -73,7 +73,7 @@ model_list: rpm: 6 router_settings: - fallbacks: [{"gpt-3.5-turbo": ["gpt-4"]}] + fallbacks: [{"gpt-4o": ["gpt-4"]}] ``` @@ -138,7 +138,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \ ### Explanation -Fallbacks are done in-order - ["gpt-3.5-turbo, "gpt-4", "gpt-4-32k"], will do 'gpt-3.5-turbo' first, then 'gpt-4', etc. +Fallbacks are done in-order - ["gpt-4o, "gpt-4", "gpt-4-32k"], will do 'gpt-4o' first, then 'gpt-4', etc. You can also set [`default_fallbacks`](#default-fallbacks), in case a specific model group is misconfigured / bad. @@ -154,10 +154,10 @@ Set fallbacks in the `.completion()` call for SDK and client-side for proxy. In this request the following will occur: 1. The request to `model="zephyr-beta"` will fail -2. litellm proxy will loop through all the model_groups specified in `fallbacks=["gpt-3.5-turbo"]` -3. The request to `model="gpt-3.5-turbo"` will succeed and the client making the request will get a response from gpt-3.5-turbo +2. litellm proxy will loop through all the model_groups specified in `fallbacks=["gpt-4o"]` +3. The request to `model="gpt-4o"` will succeed and the client making the request will get a response from gpt-4o -👉 Key Change: `"fallbacks": ["gpt-3.5-turbo"]` +👉 Key Change: `"fallbacks": ["gpt-4o"]` @@ -168,7 +168,7 @@ from litellm import Router router = Router(model_list=[..]) # defined in Step 1. resp = router.completion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{"role": "user", "content": "Hey, how's it going?"}], mock_testing_fallbacks=True, # 👈 trigger fallbacks fallbacks=[ @@ -204,7 +204,7 @@ response = client.chat.completions.create( } ], extra_body={ - "fallbacks": ["gpt-3.5-turbo"] + "fallbacks": ["gpt-4o"] } ) @@ -225,7 +225,7 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \ "content": "what llm are you" } ], - "fallbacks": ["gpt-3.5-turbo"] + "fallbacks": ["gpt-4o"] }' ``` @@ -247,7 +247,7 @@ chat = ChatOpenAI( openai_api_base="http://0.0.0.0:4000", model="zephyr-beta", extra_body={ - "fallbacks": ["gpt-3.5-turbo"] + "fallbacks": ["gpt-4o"] } ) @@ -296,7 +296,7 @@ from litellm import Router router = Router(model_list=[..]) # defined in Step 1. resp = router.completion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{"role": "user", "content": "Hey, how's it going?"}], mock_testing_fallbacks=True, # 👈 trigger fallbacks fallbacks=[ @@ -350,7 +350,7 @@ curl -L -X POST 'http://0.0.0.0:4000/v1/chat/completions' \ -H 'Content-Type: application/json' \ -H 'Authorization: Bearer sk-1234' \ -d '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", @@ -551,7 +551,7 @@ To set fallbacks, just do: ``` litellm_settings: - fallbacks: [{"zephyr-beta": ["gpt-3.5-turbo"]}] + fallbacks: [{"zephyr-beta": ["gpt-4o"]}] ``` **Covers all errors (429, 500, etc.)** @@ -571,19 +571,19 @@ model_list: litellm_params: model: huggingface/HuggingFaceH4/zephyr-7b-beta api_base: http://0.0.0.0:8003 - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: - model: gpt-3.5-turbo + model: gpt-4o api_key: - - model_name: gpt-3.5-turbo-16k + - model_name: gpt-4o-16k litellm_params: - model: gpt-3.5-turbo-16k + model: gpt-4o-16k api_key: litellm_settings: num_retries: 3 # retry call 3 times on each model_name (e.g. zephyr-beta) request_timeout: 10 # raise Timeout error if call takes longer than 10s. Sets litellm.request_timeout - fallbacks: [{"zephyr-beta": ["gpt-3.5-turbo"]}] # fallback to gpt-3.5-turbo if call fails num_retries + fallbacks: [{"zephyr-beta": ["gpt-4o"]}] # fallback to gpt-4o if call fails num_retries allowed_fails: 3 # cooldown model if it fails > 1 call in a minute. cooldown_time: 30 # how long to cooldown model if fails/min > allowed_fails ``` @@ -749,14 +749,14 @@ For azure deployments, set the base model. Pick the base model from [this list]( -Filter older instances of a model (e.g. gpt-3.5-turbo) with smaller context windows +Filter older instances of a model (e.g. gpt-4o) with smaller context windows ```yaml router_settings: enable_pre_call_checks: true # 1. Enable pre-call checks model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: model: azure/chatgpt-v-2 api_base: os.environ/AZURE_API_BASE @@ -765,9 +765,9 @@ model_list: model_info: base_model: azure/gpt-4-1106-preview # 2. 👈 (azure-only) SET BASE MODEL - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: - model: gpt-3.5-turbo-1106 + model: gpt-4o-1106 api_key: os.environ/OPENAI_API_KEY ``` @@ -792,7 +792,7 @@ text = "What is the meaning of 42?" * 5000 # request sent to model set on litellm proxy, `litellm --model` response = client.chat.completions.create( - model="gpt-3.5-turbo", + model="gpt-4o", messages = [ {"role": "system", "content": text}, {"role": "user", "content": "Who was Alexander?"}, @@ -813,7 +813,7 @@ router_settings: enable_pre_call_checks: true # 1. Enable pre-call checks model_list: - - model_name: gpt-3.5-turbo-small + - model_name: gpt-4o-small litellm_params: model: azure/chatgpt-v-2 api_base: os.environ/AZURE_API_BASE @@ -822,9 +822,9 @@ model_list: model_info: base_model: azure/gpt-4-1106-preview # 2. 👈 (azure-only) SET BASE MODEL - - model_name: gpt-3.5-turbo-large + - model_name: gpt-4o-large litellm_params: - model: gpt-3.5-turbo-1106 + model: gpt-4o-1106 api_key: os.environ/OPENAI_API_KEY - model_name: claude-opus @@ -833,7 +833,7 @@ model_list: api_key: os.environ/ANTHROPIC_API_KEY litellm_settings: - context_window_fallbacks: [{"gpt-3.5-turbo-small": ["gpt-3.5-turbo-large", "claude-opus"]}] + context_window_fallbacks: [{"gpt-4o-small": ["gpt-4o-large", "claude-opus"]}] ``` **2. Start proxy** @@ -857,7 +857,7 @@ text = "What is the meaning of 42?" * 5000 # request sent to model set on litellm proxy, `litellm --model` response = client.chat.completions.create( - model="gpt-3.5-turbo", + model="gpt-4o", messages = [ {"role": "system", "content": text}, {"role": "user", "content": "Who was Alexander?"}, @@ -877,7 +877,7 @@ Fallback across providers (e.g. from Azure OpenAI to Anthropic) if you hit conte ```yaml model_list: - - model_name: gpt-3.5-turbo-small + - model_name: gpt-4o-small litellm_params: model: azure/chatgpt-v-2 api_base: os.environ/AZURE_API_BASE @@ -890,7 +890,7 @@ model_list: api_key: os.environ/ANTHROPIC_API_KEY litellm_settings: - content_policy_fallbacks: [{"gpt-3.5-turbo-small": ["claude-opus"]}] + content_policy_fallbacks: [{"gpt-4o-small": ["claude-opus"]}] ``` @@ -902,7 +902,7 @@ You can also set default_fallbacks, in case a specific model group is misconfigu ```yaml model_list: - - model_name: gpt-3.5-turbo-small + - model_name: gpt-4o-small litellm_params: model: azure/chatgpt-v-2 api_base: os.environ/AZURE_API_BASE @@ -920,7 +920,7 @@ litellm_settings: This will default to claude-opus in case any model fails. -A model-specific fallbacks (e.g. `{"gpt-3.5-turbo-small": ["claude-opus"]}`) overrides default fallback. +A model-specific fallbacks (e.g. `{"gpt-4o-small": ["claude-opus"]}`) overrides default fallback. ### EU-Region Filtering (Pre-Call Checks) @@ -937,7 +937,7 @@ router_settings: enable_pre_call_checks: true # 1. Enable pre-call checks model_list: -- model_name: gpt-3.5-turbo +- model_name: gpt-4o litellm_params: model: azure/chatgpt-v-2 api_base: os.environ/AZURE_API_BASE @@ -945,9 +945,9 @@ model_list: api_version: "2023-07-01-preview" region_name: "eu" # 👈 SET EU-REGION -- model_name: gpt-3.5-turbo +- model_name: gpt-4o litellm_params: - model: gpt-3.5-turbo-1106 + model: gpt-4o-1106 api_key: os.environ/OPENAI_API_KEY - model_name: gemini-pro @@ -976,7 +976,7 @@ client = openai.OpenAI( # request sent to model set on litellm proxy, `litellm --model` response = client.chat.completions.with_raw_response.create( - model="gpt-3.5-turbo", + model="gpt-4o", messages = [{"role": "user", "content": "Who was Alexander?"}] ) @@ -1055,7 +1055,7 @@ curl -L -X POST 'http://0.0.0.0:4000/v1/chat/completions' \ "content": "List 5 important events in the XIX century" } ], - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "disable_fallbacks": true # 👈 DISABLE FALLBACKS }' ``` diff --git a/docs/my-website/docs/proxy/service_accounts.md b/docs/my-website/docs/proxy/service_accounts.md index 49fe0173b07..3293b59190c 100644 --- a/docs/my-website/docs/proxy/service_accounts.md +++ b/docs/my-website/docs/proxy/service_accounts.md @@ -53,7 +53,7 @@ curl --location 'http://localhost:4000/chat/completions' \ --header 'Authorization: Bearer ' \ --header 'Content-Type: application/json' \ --data '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", @@ -86,7 +86,7 @@ curl --location 'http://localhost:4000/chat/completions' \ --header 'Authorization: Bearer ' \ --header 'Content-Type: application/json' \ --data '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", @@ -115,7 +115,7 @@ Expected Response } ], "created": 1677652288, - "model": "gpt-3.5-turbo-0125", + "model": "gpt-4o-0125", "object": "chat.completion", "system_fingerprint": "fp_44709d6fcb", "usage": { diff --git a/docs/my-website/docs/proxy/tag_routing.md b/docs/my-website/docs/proxy/tag_routing.md index a1ae52e5e45..3babfe8a6b4 100644 --- a/docs/my-website/docs/proxy/tag_routing.md +++ b/docs/my-website/docs/proxy/tag_routing.md @@ -85,7 +85,7 @@ Response } ], "created": 1677652288, - "model": "gpt-3.5-turbo-0125", + "model": "gpt-4o-0125", "object": "chat.completion", "system_fingerprint": "fp_44709d6fcb", "usage": { diff --git a/docs/my-website/docs/proxy/team_based_routing.md b/docs/my-website/docs/proxy/team_based_routing.md index 4230134dd64..eaa550c33fe 100644 --- a/docs/my-website/docs/proxy/team_based_routing.md +++ b/docs/my-website/docs/proxy/team_based_routing.md @@ -16,13 +16,13 @@ Create a config.yaml with 2 model groups + connected postgres db ```yaml model_list: - - model_name: gpt-3.5-turbo-eu # 👈 Model Group 1 + - model_name: gpt-4o-eu # 👈 Model Group 1 litellm_params: model: azure/chatgpt-v-2 api_base: os.environ/AZURE_API_BASE_EU api_key: os.environ/AZURE_API_KEY_EU api_version: "2023-07-01-preview" - - model_name: gpt-3.5-turbo-worldwide # 👈 Model Group 2 + - model_name: gpt-4o-worldwide # 👈 Model Group 2 litellm_params: model: azure/chatgpt-v-2 api_base: os.environ/AZURE_API_BASE @@ -48,7 +48,7 @@ curl --location 'http://0.0.0.0:4000/team/new' \ --header 'Content-Type: application/json' \ --data '{ "team_alias": "my-new-team_4", - "model_aliases": {"gpt-3.5-turbo": "gpt-3.5-turbo-eu"} + "model_aliases": {"gpt-4o": "gpt-4o-eu"} }' # Returns team_id: my-team-id @@ -72,7 +72,7 @@ curl --location 'http://0.0.0.0:4000/v1/chat/completions' \ --header 'Content-Type: application/json' \ --header 'Authorization: Bearer sk-A1L0C3Px2LJl53sF_kTF9A' \ --data '{ - "model": "gpt-3.5-turbo", # 👈 MODEL + "model": "gpt-4o", # 👈 MODEL "messages": [{"role": "system", "content": "You'\''re an expert at writing poems"}, {"role": "user", "content": "Write me a poem"}, {"role": "user", "content": "What'\''s your name?"}], "user": "usha" }' diff --git a/docs/my-website/docs/proxy/timeout.md b/docs/my-website/docs/proxy/timeout.md index 52cb160cf76..9c5459d996e 100644 --- a/docs/my-website/docs/proxy/timeout.md +++ b/docs/my-website/docs/proxy/timeout.md @@ -55,7 +55,7 @@ from litellm import Router import asyncio model_list = [{ - "model_name": "gpt-3.5-turbo", + "model_name": "gpt-4o", "litellm_params": { "model": "azure/chatgpt-v-2", "api_key": os.getenv("AZURE_API_KEY"), @@ -70,7 +70,7 @@ model_list = [{ router = Router(model_list=model_list, routing_strategy="least-busy") async def router_acompletion(): response = await router.acompletion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{"role": "user", "content": "Hey, how's it going?"}] ) print(response) @@ -84,7 +84,7 @@ asyncio.run(router_acompletion()) ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: model: azure/gpt-turbo-small-eu api_base: https://my-endpoint-europe-berri-992.openai.azure.com/ @@ -92,7 +92,7 @@ model_list: timeout: 0.1 # timeout in (seconds) stream_timeout: 0.01 # timeout for stream requests (seconds) max_retries: 5 - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: model: azure/gpt-turbo-small-ca api_base: https://my-endpoint-canada-berri992.openai.azure.com/ @@ -130,7 +130,7 @@ model_list = [{...}] router = Router(model_list=model_list) response = router.completion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{"role": "user", "content": "what color is red"}], timeout=1 ) @@ -146,7 +146,7 @@ response = router.completion( curl --location 'http://0.0.0.0:4000/chat/completions' \ --header 'Content-Type: application/json' \ --data-raw '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ {"role": "user", "content": "what color is red"} ], @@ -167,7 +167,7 @@ client = openai.OpenAI( ) response = client.chat.completions.create( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[ {"role": "user", "content": "what color is red"} ], diff --git a/docs/my-website/docs/proxy/token_auth.md b/docs/my-website/docs/proxy/token_auth.md index 7364ae0fb56..fbf507f7101 100644 --- a/docs/my-website/docs/proxy/token_auth.md +++ b/docs/my-website/docs/proxy/token_auth.md @@ -864,9 +864,9 @@ model_list: litellm_params: model: anthropic/claude-3-5-sonnet api_key: os.environ/ANTHROPIC_API_KEY - - model_name: gpt-3.5-turbo-testing + - model_name: gpt-4o-testing litellm_params: - model: gpt-3.5-turbo + model: gpt-4o api_key: os.environ/OPENAI_API_KEY general_settings: @@ -878,7 +878,7 @@ general_settings: - scope: litellm.api.consumer models: ["anthropic-claude"] - scope: litellm.api.gpt_3_5_turbo - models: ["gpt-3.5-turbo-testing"] + models: ["gpt-4o-testing"] enforce_scope_based_access: true # 👈 enforce scope-based access control enforce_rbac: true # 👈 enforces only a Team/User/ProxyAdmin can access the proxy. ``` @@ -905,7 +905,7 @@ curl -L -X POST 'http://0.0.0.0:4000/v1/chat/completions' \ -H 'Content-Type: application/json' \ -H 'Authorization: Bearer eyJhbGci...' \ -d '{ - "model": "gpt-3.5-turbo-testing", + "model": "gpt-4o-testing", "messages": [ { "role": "user", diff --git a/docs/my-website/docs/proxy/user_keys.md b/docs/my-website/docs/proxy/user_keys.md index 72ec8ccd759..315313c43cf 100644 --- a/docs/my-website/docs/proxy/user_keys.md +++ b/docs/my-website/docs/proxy/user_keys.md @@ -67,7 +67,7 @@ client = openai.OpenAI( # request sent to model set on litellm proxy, `litellm --model` response = client.chat.completions.create( - model="gpt-3.5-turbo", + model="gpt-4o", messages = [ { "role": "user", @@ -105,7 +105,7 @@ client = openai.AzureOpenAI( # request sent to model set on litellm proxy, `litellm --model` response = client.chat.completions.create( - model="gpt-3.5-turbo", + model="gpt-4o", messages = [ { "role": "user", @@ -169,7 +169,7 @@ Pass `metadata` as part of the request body curl --location 'http://0.0.0.0:4000/chat/completions' \ --header 'Content-Type: application/json' \ --data '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", @@ -201,7 +201,7 @@ os.environ["OPENAI_API_KEY"] = "anything" chat = ChatOpenAI( openai_api_base="http://0.0.0.0:4000", - model = "gpt-3.5-turbo", + model = "gpt-4o", temperature=0.1, extra_body={ "metadata": { @@ -261,7 +261,7 @@ const openai = new OpenAI({ async function main() { const chatCompletion = await openai.chat.completions.create({ messages: [{ role: 'user', content: 'Say this is a test' }], - model: 'gpt-3.5-turbo', + model: 'gpt-4o', }, {"metadata": { "generation_name": "ishaan-generation-openaijs-client", "generation_id": "openaijs-client-gen-id22", @@ -372,7 +372,7 @@ client = openai.OpenAI( ) response = client.chat.completions.create( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{"role": "user", "content": "Hello!"}], extra_body={ "metadata": { @@ -414,7 +414,7 @@ response = chat.invoke([HumanMessage(content="Generate a blog post")]) curl --location 'http://0.0.0.0:4000/chat/completions' \ --header 'Content-Type: application/json' \ --data '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [{"role": "user", "content": "Hello!"}], "metadata": { "tags": ["api-test", "development"], @@ -438,7 +438,7 @@ const openai = new OpenAI({ async function main() { const response = await openai.chat.completions.create({ messages: [{ role: 'user', content: 'Hello!' }], - model: 'gpt-3.5-turbo', + model: 'gpt-4o', metadata: { tags: ["javascript-client", "api-test"], trace_user_id: "js-user-789" @@ -815,7 +815,7 @@ client = openai.OpenAI( ) # request sent to model set on litellm proxy, `litellm --model` -response = client.chat.completions.create(model="gpt-3.5-turbo", messages = [ +response = client.chat.completions.create(model="gpt-4o", messages = [ { "role": "user", "content": "this is a test request, write a short poem" @@ -830,7 +830,7 @@ print(response) #### Start the LiteLLM proxy ```shell -litellm --model gpt-3.5-turbo +litellm --model gpt-4o #INFO: Proxy running on http://0.0.0.0:4000 ``` @@ -972,11 +972,11 @@ Use this when you want to send 1 request to N Models #### Expected Request Format -Pass model as a string of comma separated value of models. Example `"model"="llama3,gpt-3.5-turbo"` +Pass model as a string of comma separated value of models. Example `"model"="llama3,gpt-4o"` This same request will be sent to the following model groups on the [litellm proxy config.yaml](https://docs.litellm.ai/docs/proxy/configs) - `model_name="llama3"` -- `model_name="gpt-3.5-turbo"` +- `model_name="gpt-4o"` @@ -989,7 +989,7 @@ import openai client = openai.OpenAI(api_key="sk-1234", base_url="http://0.0.0.0:4000") response = client.chat.completions.create( - model="gpt-3.5-turbo,llama3", + model="gpt-4o,llama3", messages=[ {"role": "user", "content": "this is a test request, write a short poem"} ], @@ -1022,7 +1022,7 @@ Get a list of responses when `model` is passed as a list ) ], created=1715462919, - model='gpt-3.5-turbo-0125', + model='gpt-4o-0125', object='chat.completion', system_fingerprint=None, usage=CompletionUsage( @@ -1072,7 +1072,7 @@ curl --location 'http://localhost:4000/chat/completions' \ --header 'Authorization: Bearer sk-1234' \ --header 'Content-Type: application/json' \ --data '{ - "model": "llama3,gpt-3.5-turbo", + "model": "llama3,gpt-4o", "max_tokens": 10, "user": "litellm2", "messages": [ @@ -1128,7 +1128,7 @@ Get a list of responses when `model` is passed as a list } ], "created": 1715459877, - "model": "gpt-3.5-turbo-0125", + "model": "gpt-4o-0125", "object": "chat.completion", "system_fingerprint": null, "usage": { diff --git a/docs/my-website/docs/proxy/user_onboarding.md b/docs/my-website/docs/proxy/user_onboarding.md index ecbdc11db43..29634fd3903 100644 --- a/docs/my-website/docs/proxy/user_onboarding.md +++ b/docs/my-website/docs/proxy/user_onboarding.md @@ -63,7 +63,7 @@ curl -X POST http://localhost:4000/v1/chat/completions \ -H "Authorization: Bearer " \ -H "Content-Type: application/json" \ -d '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [{"role": "user", "content": "Hello!"}] }' ``` diff --git a/docs/my-website/docs/proxy/users.md b/docs/my-website/docs/proxy/users.md index 88a7a0f1e07..4441647c86a 100644 --- a/docs/my-website/docs/proxy/users.md +++ b/docs/my-website/docs/proxy/users.md @@ -51,7 +51,7 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \ --header 'Autherization: Bearer sk-1234' \ --header 'Content-Type: application/json' \ --data '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", @@ -704,8 +704,8 @@ curl --location 'http://0.0.0.0:4000/team/new' \ --header 'Content-Type: application/json' \ --data '{ "team_id": "my-prod-team", - "model_rpm_limit": {"gpt-4": 100, "gpt-3.5-turbo": 200}, - "model_tpm_limit": {"gpt-4": 10000, "gpt-3.5-turbo": 20000} + "model_rpm_limit": {"gpt-4": 100, "gpt-4o": 200}, + "model_tpm_limit": {"gpt-4": 10000, "gpt-4o": 20000} }' ``` @@ -717,8 +717,8 @@ curl --location 'http://0.0.0.0:4000/team/update' \ --header 'Content-Type: application/json' \ --data '{ "team_id": "my-prod-team", - "model_rpm_limit": {"gpt-4": 100, "gpt-3.5-turbo": 200}, - "model_tpm_limit": {"gpt-4": 10000, "gpt-3.5-turbo": 20000} + "model_rpm_limit": {"gpt-4": 100, "gpt-4o": 200}, + "model_tpm_limit": {"gpt-4": 10000, "gpt-4o": 20000} }' ``` @@ -733,8 +733,8 @@ curl --location 'http://0.0.0.0:4000/team/update' \ --data '{ "team_id": "my-prod-team", "metadata": { - "model_rpm_limit": {"gpt-4": 100, "gpt-3.5-turbo": 200}, - "model_tpm_limit": {"gpt-4": 10000, "gpt-3.5-turbo": 20000} + "model_rpm_limit": {"gpt-4": 100, "gpt-4o": 200}, + "model_tpm_limit": {"gpt-4": 10000, "gpt-4o": 20000} } }' ``` @@ -948,9 +948,9 @@ This will NOT apply if a key has a team_id (team budgets will apply then). [Tell ```yaml model_list: - - model_name: "gpt-3.5-turbo" + - model_name: "gpt-4o" litellm_params: - model: gpt-3.5-turbo + model: gpt-4o api_key: os.environ/OPENAI_API_KEY litellm_settings: @@ -983,7 +983,7 @@ curl -L -X POST 'http://0.0.0.0:4000/chat/completions' \ -H 'Content-Type: application/json' \ -H 'Authorization: Bearer sk-X53RdxnDhzamRwjKXR4IHg' \ -d '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [{"role": "user", "content": "Hey, how's it going?"}] }' ``` diff --git a/docs/my-website/docs/proxy/virtual_keys.md b/docs/my-website/docs/proxy/virtual_keys.md index c74aa75ff4a..9d20cb39282 100644 --- a/docs/my-website/docs/proxy/virtual_keys.md +++ b/docs/my-website/docs/proxy/virtual_keys.md @@ -43,7 +43,7 @@ model_list: - model_name: gpt-4 litellm_params: model: ollama/llama2 - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: model: ollama/llama2 @@ -64,7 +64,7 @@ litellm --config /path/to/config.yaml curl 'http://0.0.0.0:4000/key/generate' \ --header 'Authorization: Bearer ' \ --header 'Content-Type: application/json' \ ---data-raw '{"models": ["gpt-3.5-turbo", "gpt-4"], "metadata": {"user": "ishaan@berri.ai"}}' +--data-raw '{"models": ["gpt-4o", "gpt-4"], "metadata": {"user": "ishaan@berri.ai"}}' ``` ## Spend Tracking @@ -106,12 +106,12 @@ This is automatically updated (in USD) when calls are made to /completions, /cha "spend": 0.0001065, # 👈 SPEND "expires": "2023-11-24T23:19:11.131000Z", "models": [ - "gpt-3.5-turbo", + "gpt-4o", "gpt-4", "claude-2" ], "aliases": { - "mistral-7b": "gpt-3.5-turbo" + "mistral-7b": "gpt-4o" }, "config": {} } @@ -147,7 +147,7 @@ curl --location 'http://localhost:4000/user/new' \ curl 'http://0.0.0.0:4000/key/generate' \ --header 'Authorization: Bearer ' \ --header 'Content-Type: application/json' \ ---data-raw '{"models": ["gpt-3.5-turbo", "gpt-4"], "user_id": "my-unique-id"}' +--data-raw '{"models": ["gpt-4o", "gpt-4"], "user_id": "my-unique-id"}' ``` Returns a key - `sk-...`. @@ -200,7 +200,7 @@ curl --location 'http://localhost:4000/team/new' \ curl 'http://0.0.0.0:4000/key/generate' \ --header 'Authorization: Bearer ' \ --header 'Content-Type: application/json' \ ---data-raw '{"models": ["gpt-3.5-turbo", "gpt-4"], "team_id": "my-unique-id"}' +--data-raw '{"models": ["gpt-4o", "gpt-4"], "team_id": "my-unique-id"}' ``` Returns a key - `sk-...`. @@ -265,7 +265,7 @@ curl -X POST "https://0.0.0.0:4000/key/generate" \ -H "Content-Type: application/json" \ -d '{ "models": ["my-free-tier"], - "aliases": {"gpt-3.5-turbo": "my-free-tier"}, # 👈 KEY CHANGE + "aliases": {"gpt-4o": "my-free-tier"}, # 👈 KEY CHANGE "duration": "30min" }' ``` @@ -279,7 +279,7 @@ curl -X POST "https://0.0.0.0:4000/key/generate" \ -H "Authorization: Bearer " \ -H "Content-Type: application/json" \ -d '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", @@ -336,7 +336,7 @@ curl http://localhost:4000/v1/chat/completions \ Expect to see a successful response from the litellm proxy since the key passed in `X-Litellm-Key` is valid ```shell -{"id":"chatcmpl-f9b2b79a7c30477ab93cd0e717d1773e","choices":[{"finish_reason":"stop","index":0,"message":{"content":"\n\nHello there, how may I assist you today?","role":"assistant","tool_calls":null,"function_call":null}}],"created":1677652288,"model":"gpt-3.5-turbo-0125","object":"chat.completion","system_fingerprint":"fp_44709d6fcb","usage":{"completion_tokens":12,"prompt_tokens":9,"total_tokens":21} +{"id":"chatcmpl-f9b2b79a7c30477ab93cd0e717d1773e","choices":[{"finish_reason":"stop","index":0,"message":{"content":"\n\nHello there, how may I assist you today?","role":"assistant","tool_calls":null,"function_call":null}}],"created":1677652288,"model":"gpt-4o-0125","object":"chat.completion","system_fingerprint":"fp_44709d6fcb","usage":{"completion_tokens":12,"prompt_tokens":9,"total_tokens":21} ``` @@ -473,7 +473,7 @@ e.g. if they're both in the same dir - `./config.yaml` and `./custom_auth.py`, t model_list: - model_name: "openai-model" litellm_params: - model: "gpt-3.5-turbo" + model: "gpt-4o" litellm_settings: drop_params: True @@ -548,7 +548,7 @@ curl 'http://localhost:4000/key/sk-1234/regenerate' \ }, "models": [ "gpt-4", - "gpt-3.5-turbo" + "gpt-4o" ], "grace_period": "48h" }'