diff --git a/docs/my-website/docs/proxy/caching.md b/docs/my-website/docs/proxy/caching.md index 3357dcb28b2..74512e6fd60 100644 --- a/docs/my-website/docs/proxy/caching.md +++ b/docs/my-website/docs/proxy/caching.md @@ -34,9 +34,9 @@ Caching can be enabled by adding the `cache` key in the `config.yaml` ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: - model: gpt-3.5-turbo + model: gpt-4o - model_name: text-embedding-ada-002 litellm_params: model: text-embedding-ada-002 @@ -382,9 +382,9 @@ one** ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: - model: gpt-3.5-turbo + model: gpt-4o - model_name: text-embedding-ada-002 litellm_params: model: text-embedding-ada-002 @@ -415,9 +415,9 @@ $ litellm --config /path/to/config.yaml ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: - model: gpt-3.5-turbo + model: gpt-4o - model_name: text-embedding-ada-002 litellm_params: model: text-embedding-ada-002 @@ -457,9 +457,9 @@ Caching can be enabled by adding the `cache` key in the `config.yaml` ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: - model: gpt-3.5-turbo + model: gpt-4o - model_name: azure-embedding-model litellm_params: model: azure/azure-embedding-model @@ -558,7 +558,7 @@ Send the same request twice: curl http://0.0.0.0:4000/v1/chat/completions \ -H "Content-Type: application/json" \ -d '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [{"role": "user", "content": "write a poem about litellm!"}], "temperature": 0.7 }' @@ -566,7 +566,7 @@ curl http://0.0.0.0:4000/v1/chat/completions \ curl http://0.0.0.0:4000/v1/chat/completions \ -H "Content-Type: application/json" \ -d '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [{"role": "user", "content": "write a poem about litellm!"}], "temperature": 0.7 }' @@ -625,7 +625,7 @@ client = OpenAI( chat_completion = client.chat.completions.create( messages=[{"role": "user", "content": "Hello"}], - model="gpt-3.5-turbo", + model="gpt-4o", extra_body={ "cache": { "ttl": 300 # Cache response for 5 minutes @@ -643,7 +643,7 @@ curl http://localhost:4000/v1/chat/completions \ -H "Content-Type: application/json" \ -H "Authorization: Bearer sk-1234" \ -d '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "cache": {"ttl": 300}, "messages": [ {"role": "user", "content": "Hello"} @@ -671,7 +671,7 @@ client = OpenAI( chat_completion = client.chat.completions.create( messages=[{"role": "user", "content": "Hello"}], - model="gpt-3.5-turbo", + model="gpt-4o", extra_body={ "cache": { "s-maxage": 600 # Only use cache if less than 10 minutes old @@ -689,7 +689,7 @@ curl http://localhost:4000/v1/chat/completions \ -H "Content-Type: application/json" \ -H "Authorization: Bearer sk-1234" \ -d '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "cache": {"s-maxage": 600}, "messages": [ {"role": "user", "content": "Hello"} @@ -717,7 +717,7 @@ client = OpenAI( chat_completion = client.chat.completions.create( messages=[{"role": "user", "content": "Hello"}], - model="gpt-3.5-turbo", + model="gpt-4o", extra_body={ "cache": { "no-cache": True # Skip cache check, get fresh response @@ -735,7 +735,7 @@ curl http://localhost:4000/v1/chat/completions \ -H "Content-Type: application/json" \ -H "Authorization: Bearer sk-1234" \ -d '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "cache": {"no-cache": true}, "messages": [ {"role": "user", "content": "Hello"} @@ -763,7 +763,7 @@ client = OpenAI( chat_completion = client.chat.completions.create( messages=[{"role": "user", "content": "Hello"}], - model="gpt-3.5-turbo", + model="gpt-4o", extra_body={ "cache": { "no-store": True # Don't cache this response @@ -781,7 +781,7 @@ curl http://localhost:4000/v1/chat/completions \ -H "Content-Type: application/json" \ -H "Authorization: Bearer sk-1234" \ -d '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "cache": {"no-store": true}, "messages": [ {"role": "user", "content": "Hello"} @@ -809,7 +809,7 @@ client = OpenAI( chat_completion = client.chat.completions.create( messages=[{"role": "user", "content": "Hello"}], - model="gpt-3.5-turbo", + model="gpt-4o", extra_body={ "cache": { "namespace": "my-custom-namespace" # Store in custom namespace @@ -827,7 +827,7 @@ curl http://localhost:4000/v1/chat/completions \ -H "Content-Type: application/json" \ -H "Authorization: Bearer sk-1234" \ -d '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "cache": {"namespace": "my-custom-namespace"}, "messages": [ {"role": "user", "content": "Hello"} @@ -908,9 +908,9 @@ litellm_settings: ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: - model: gpt-3.5-turbo + model: gpt-4o - model_name: text-embedding-ada-002 litellm_params: model: text-embedding-ada-002 @@ -956,7 +956,7 @@ curl -i --location 'http://0.0.0.0:4000/chat/completions' \ --header 'Authorization: Bearer sk-1234' \ --header 'Content-Type: application/json' \ --data '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "user": "ishan", "messages": [ { @@ -1029,7 +1029,7 @@ chat_completion = client.chat.completions.create( "content": "Say this is a test", } ], - model="gpt-3.5-turbo", + model="gpt-4o", extra_body = { # OpenAI python accepts extra args in extra_body "cache": {"use-cache": True} } @@ -1045,7 +1045,7 @@ curl http://localhost:4000/v1/chat/completions \ -H "Content-Type: application/json" \ -H "Authorization: Bearer sk-1234" \ -d '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "cache": {"use-cache": True} "messages": [ {"role": "user", "content": "Say this is a test"} diff --git a/docs/my-website/docs/proxy/call_hooks.md b/docs/my-website/docs/proxy/call_hooks.md index 5935a29c50b..850978e00b1 100644 --- a/docs/my-website/docs/proxy/call_hooks.md +++ b/docs/my-website/docs/proxy/call_hooks.md @@ -135,9 +135,9 @@ proxy_handler_instance = MyCustomHandler() ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: - model: gpt-3.5-turbo + model: gpt-4o litellm_settings: callbacks: custom_callbacks.proxy_handler_instance # sets litellm.callbacks = [proxy_handler_instance] @@ -151,7 +151,7 @@ $ litellm /path/to/config.yaml ```shell curl --location 'http://0.0.0.0:4000/chat/completions' \ --data ' { - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", @@ -231,9 +231,9 @@ proxy_handler_instance = MyCustomHandler() ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: - model: gpt-3.5-turbo + model: gpt-4o litellm_settings: callbacks: custom_callbacks.proxy_handler_instance # sets litellm.callbacks = [proxy_handler_instance] @@ -247,7 +247,7 @@ $ litellm /path/to/config.yaml ```shell curl --location 'http://0.0.0.0:4000/chat/completions' \ --data ' { - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", @@ -318,9 +318,9 @@ proxy_handler_instance = MyCustomHandler() ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: - model: gpt-3.5-turbo + model: gpt-4o litellm_settings: callbacks: custom_callbacks.proxy_handler_instance # sets litellm.callbacks = [proxy_handler_instance] @@ -335,7 +335,7 @@ $ litellm /path/to/config.yaml ```shell curl --location 'http://0.0.0.0:4000/chat/completions' \ --data ' { - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", diff --git a/docs/my-website/docs/proxy/clientside_auth.md b/docs/my-website/docs/proxy/clientside_auth.md index c696737adc0..8b258f3c967 100644 --- a/docs/my-website/docs/proxy/clientside_auth.md +++ b/docs/my-website/docs/proxy/clientside_auth.md @@ -41,7 +41,7 @@ user_config = { { 'model_name': 'user-openai-instance', 'litellm_params': { - 'model': 'gpt-3.5-turbo', + 'model': 'gpt-4o', 'api_key': os.getenv('OPENAI_API_KEY'), 'timeout': 10, }, @@ -109,7 +109,7 @@ const userConfig = { { model_name: 'user-openai-instance', litellm_params: { - model: 'gpt-3.5-turbo', + model: 'gpt-4o', api_key: process.env.OPENAI_API_KEY, timeout: 10, }, @@ -140,7 +140,7 @@ const openai = new OpenAI({ async function main() { const chatCompletion = await openai.chat.completions.create({ messages: [{ role: 'user', content: 'Say this is a test' }], - model: 'gpt-3.5-turbo', + model: 'gpt-4o', user_config: userConfig // # 👈 User config }); } @@ -188,7 +188,7 @@ client = openai.OpenAI( ) # request sent to model set on litellm proxy, `litellm --model` -response = client.chat.completions.create(model="gpt-3.5-turbo", messages = [ +response = client.chat.completions.create(model="gpt-4o", messages = [ { "role": "user", "content": "this is a test request, write a short poem" @@ -213,7 +213,7 @@ client = openai.OpenAI( ) # request sent to model set on litellm proxy, `litellm --model` -response = client.chat.completions.create(model="gpt-3.5-turbo", messages = [ +response = client.chat.completions.create(model="gpt-4o", messages = [ { "role": "user", "content": "this is a test request, write a short poem" @@ -245,7 +245,7 @@ const openai = new OpenAI({ async function main() { const chatCompletion = await openai.chat.completions.create({ messages: [{ role: 'user', content: 'Say this is a test' }], - model: 'gpt-3.5-turbo', + model: 'gpt-4o', api_key: "my-bad-key" // 👈 User Key }); } @@ -272,7 +272,7 @@ client = openai.OpenAI( # request sent to model set on litellm proxy, `litellm --model` response = client.chat.completions.create( - model="gpt-3.5-turbo", + model="gpt-4o", messages = [ { "role": "user", diff --git a/docs/my-website/docs/proxy/config_settings.md b/docs/my-website/docs/proxy/config_settings.md index d7542fc2c3d..c59575a0d1d 100644 --- a/docs/my-website/docs/proxy/config_settings.md +++ b/docs/my-website/docs/proxy/config_settings.md @@ -34,8 +34,8 @@ litellm_settings: # Fallbacks, reliability default_fallbacks: ["claude-opus"] # set default_fallbacks, in case a specific model group is misconfigured / bad. - content_policy_fallbacks: [{ "gpt-3.5-turbo-small": ["claude-opus"] }] # fallbacks for ContentPolicyErrors - context_window_fallbacks: [{ "gpt-3.5-turbo-small": ["gpt-3.5-turbo-large", "claude-opus"] }] # fallbacks for ContextWindowExceededErrors + content_policy_fallbacks: [{ "gpt-4o-small": ["claude-opus"] }] # fallbacks for ContentPolicyErrors + context_window_fallbacks: [{ "gpt-4o-small": ["gpt-4o-large", "claude-opus"] }] # fallbacks for ContextWindowExceededErrors # MCP Aliases - Map aliases to MCP server names for easier tool access mcp_aliases: { @@ -349,7 +349,7 @@ router_settings: | model_group_alias | dict | Model group alias mapping. E.g. `{"claude-3-haiku": "claude-3-haiku-20240229"}` | | num_retries | int | Number of retries for a request. Defaults to 3. | | default_fallbacks | Optional[List[str]] | Fallbacks to try if no model group-specific fallbacks are defined. | -| caching_groups | Optional[List[tuple]] | List of model groups for caching across model groups. Defaults to None. - e.g. caching_groups=[("openai-gpt-3.5-turbo", "azure-gpt-3.5-turbo")]| +| caching_groups | Optional[List[tuple]] | List of model groups for caching across model groups. Defaults to None. - e.g. caching_groups=[("openai-gpt-4o", "azure-gpt-4o")]| | alerting_config | AlertingConfig | [SDK-only arg] Slack alerting configuration. Defaults to None. [Further Docs](../routing.md#alerting-) | | assistants_config | AssistantsConfig | Set on proxy via `assistant_settings`. [Further docs](../assistants.md) | | set_verbose | boolean | [DEPRECATED PARAM - see debug docs](./debugging) If true, sets the logging level to verbose. | diff --git a/docs/my-website/docs/proxy/configs.md b/docs/my-website/docs/proxy/configs.md index 84a6fac1210..cc054c2beb2 100644 --- a/docs/my-website/docs/proxy/configs.md +++ b/docs/my-website/docs/proxy/configs.md @@ -400,9 +400,9 @@ model_list: model: gpt-4o api_key: rpm: 200 - - model_name: gpt-3.5-turbo-16k + - model_name: gpt-4o-16k litellm_params: - model: gpt-3.5-turbo-16k + model: gpt-4o-16k api_key: rpm: 100 @@ -410,7 +410,7 @@ litellm_settings: num_retries: 3 # retry call 3 times on each model_name (e.g. zephyr-beta) request_timeout: 10 # raise Timeout error if call takes longer than 10s. Sets litellm.request_timeout fallbacks: [{"zephyr-beta": ["gpt-4o"]}] # fallback to gpt-4o if call fails num_retries - context_window_fallbacks: [{"zephyr-beta": ["gpt-3.5-turbo-16k"]}, {"gpt-4o": ["gpt-3.5-turbo-16k"]}] # fallback to gpt-3.5-turbo-16k if context window error + context_window_fallbacks: [{"zephyr-beta": ["gpt-4o-16k"]}, {"gpt-4o": ["gpt-4o-16k"]}] # fallback to gpt-4o-16k if context window error allowed_fails: 3 # cooldown model if it fails > 1 call in a minute. router_settings: # router_settings are optional @@ -497,9 +497,9 @@ Supported Environments: 2. For each model set the list of supported environments in `model_info.supported_environments` ```yaml model_list: - - model_name: gpt-3.5-turbo-16k + - model_name: gpt-4o-16k litellm_params: - model: openai/gpt-3.5-turbo-16k + model: openai/gpt-4o-16k api_key: os.environ/OPENAI_API_KEY model_info: supported_environments: ["development", "production", "staging"] diff --git a/docs/my-website/docs/proxy/cost_tracking.md b/docs/my-website/docs/proxy/cost_tracking.md index f28eec287d4..d60af663fa8 100644 --- a/docs/my-website/docs/proxy/cost_tracking.md +++ b/docs/my-website/docs/proxy/cost_tracking.md @@ -388,7 +388,7 @@ client = openai.OpenAI( response = client.chat.completions.create( - model="gpt-3.5-turbo", + model="gpt-4o", messages = [ { "role": "user", @@ -420,7 +420,7 @@ async function runOpenAI() { try { const response = await client.chat.completions.create({ - model: "gpt-3.5-turbo", + model: "gpt-4o", messages: [ { role: "user", @@ -452,7 +452,7 @@ Pass `metadata` as part of the request body curl --location 'http://0.0.0.0:4000/chat/completions' \ --header 'Content-Type: application/json' \ --data '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", @@ -477,7 +477,7 @@ from langchain.schema import HumanMessage, SystemMessage chat = ChatOpenAI( openai_api_base="http://0.0.0.0:4000", - model = "gpt-3.5-turbo", + model = "gpt-4o", temperature=0.1, extra_body={ "metadata": { @@ -574,7 +574,7 @@ curl -X GET 'http://localhost:4000/global/spend/report?start_date=2024-04-01&end "api_key": "898c28.." # the hashed api key }, { - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "spend": 0.0000825, "total_tokens": 85, "api_key": "84dc28.." # the hashed api key @@ -627,22 +627,22 @@ Output from script # Date: 2024-05-11T00:00:00+00:00 # Team: local_test_team # Total Spend: 0.003675099999999999 -# Metadata: [{'model': 'gpt-3.5-turbo', 'spend': 0.003675099999999999, 'api_key': 'b94d5e0bc3a71a573917fe1335dc0c14728c7016337451af9714924ff3a729db', 'total_tokens': 3105}] +# Metadata: [{'model': 'gpt-4o', 'spend': 0.003675099999999999, 'api_key': 'b94d5e0bc3a71a573917fe1335dc0c14728c7016337451af9714924ff3a729db', 'total_tokens': 3105}] # Date: 2024-05-13T00:00:00+00:00 # Team: Unassigned Team # Total Spend: 3.4e-05 -# Metadata: [{'model': 'gpt-3.5-turbo', 'spend': 3.4e-05, 'api_key': '9569d13c9777dba68096dea49b0b03e0aaf4d2b65d4030eda9e8a2733c3cd6e0', 'total_tokens': 50}] +# Metadata: [{'model': 'gpt-4o', 'spend': 3.4e-05, 'api_key': '9569d13c9777dba68096dea49b0b03e0aaf4d2b65d4030eda9e8a2733c3cd6e0', 'total_tokens': 50}] # Date: 2024-05-13T00:00:00+00:00 # Team: central # Total Spend: 0.000684 -# Metadata: [{'model': 'gpt-3.5-turbo', 'spend': 0.000684, 'api_key': '0323facdf3af551594017b9ef162434a9b9a8ca1bbd9ccbd9d6ce173b1015605', 'total_tokens': 498}] +# Metadata: [{'model': 'gpt-4o', 'spend': 0.000684, 'api_key': '0323facdf3af551594017b9ef162434a9b9a8ca1bbd9ccbd9d6ce173b1015605', 'total_tokens': 498}] # Date: 2024-05-13T00:00:00+00:00 # Team: local_test_team # Total Spend: 0.0005715000000000001 -# Metadata: [{'model': 'gpt-3.5-turbo', 'spend': 0.0005715000000000001, 'api_key': 'b94d5e0bc3a71a573917fe1335dc0c14728c7016337451af9714924ff3a729db', 'total_tokens': 423}] +# Metadata: [{'model': 'gpt-4o', 'spend': 0.0005715000000000001, 'api_key': 'b94d5e0bc3a71a573917fe1335dc0c14728c7016337451af9714924ff3a729db', 'total_tokens': 423}] ``` @@ -700,7 +700,7 @@ curl -X GET 'http://localhost:4000/global/spend/report?start_date=2024-04-01&end "api_key": "898c28.." # the hashed api key }, { - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "spend": 0.0000825, "total_tokens": 85, "api_key": "84dc28.." # the hashed api key @@ -778,7 +778,7 @@ curl -X GET 'http://localhost:4000/global/spend/report?start_date=2024-04-01&end "total_output_tokens": 872.0, "model_details": [ { - "model": "gpt-3.5-turbo-instruct", + "model": "gpt-4o-instruct", "total_cost": 5.85e-05, "total_input_tokens": 15, "total_output_tokens": 18 @@ -798,7 +798,7 @@ curl -X GET 'http://localhost:4000/global/spend/report?start_date=2024-04-01&end "total_output_tokens": 27.0, "model_details": [ { - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "total_cost": 5.2499999999999995e-05, "total_input_tokens": 24, "total_output_tokens": 27 @@ -931,7 +931,7 @@ client = openai.OpenAI( # request sent to model set on litellm proxy, `litellm --model` response = client.chat.completions.create( - model="gpt-3.5-turbo", + model="gpt-4o", messages = [ { "role": "user", @@ -961,7 +961,7 @@ client = openai.OpenAI( # Pass spend logs metadata via headers response = client.chat.completions.create( - model="gpt-3.5-turbo", + model="gpt-4o", messages = [ { "role": "user", @@ -992,7 +992,7 @@ async function runOpenAI() { try { const response = await client.chat.completions.create({ - model: 'gpt-3.5-turbo', + model: 'gpt-4o', messages: [ { role: 'user', @@ -1029,7 +1029,7 @@ async function runOpenAI() { try { const response = await client.chat.completions.create({ - model: 'gpt-3.5-turbo', + model: 'gpt-4o', messages: [ { role: 'user', @@ -1062,7 +1062,7 @@ Pass `metadata` as part of the request body curl --location 'http://0.0.0.0:4000/chat/completions' \ --header 'Content-Type: application/json' \ --data '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", @@ -1089,7 +1089,7 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \ --header 'Authorization: Bearer sk-1234' \ --header 'x-litellm-spend-logs-metadata: {"user_id": "12345", "project_id": "proj_abc", "request_type": "chat_completion"}' \ --data '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", @@ -1113,7 +1113,7 @@ from langchain.schema import HumanMessage, SystemMessage chat = ChatOpenAI( openai_api_base="http://0.0.0.0:4000", - model = "gpt-3.5-turbo", + model = "gpt-4o", temperature=0.1, extra_body={ "metadata": { diff --git a/docs/my-website/docs/proxy/custom_sso.md b/docs/my-website/docs/proxy/custom_sso.md index 41ecde6e369..25bfa40cc8a 100644 --- a/docs/my-website/docs/proxy/custom_sso.md +++ b/docs/my-website/docs/proxy/custom_sso.md @@ -81,7 +81,7 @@ custom_ui_sso_sign_in_handler = MyCustomSSOLoginHandler() model_list: - model_name: "openai-model" litellm_params: - model: "gpt-3.5-turbo" + model: "gpt-4o" general_settings: custom_ui_sso_sign_in_handler: custom_sso_handler.custom_ui_sso_sign_in_handler @@ -184,7 +184,7 @@ e.g. if they're both in the same dir - `./config.yaml` and `./custom_sso.py`, th model_list: - model_name: "openai-model" litellm_params: - model: "gpt-3.5-turbo" + model: "gpt-4o" general_settings: custom_sso: custom_sso.custom_sso_handler diff --git a/docs/my-website/docs/proxy/customer_usage.md b/docs/my-website/docs/proxy/customer_usage.md index 5a6c06fdc81..5d9d46c6828 100644 --- a/docs/my-website/docs/proxy/customer_usage.md +++ b/docs/my-website/docs/proxy/customer_usage.md @@ -36,7 +36,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \ --header 'Content-Type: application/json' \ --header 'Authorization: Bearer sk-1234' \ --data '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "user": "customer-123", "messages": [ { @@ -64,7 +64,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \ --header 'Authorization: Bearer sk-1234' \ --header 'x-litellm-customer-id: customer-123' \ --data '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", diff --git a/docs/my-website/docs/proxy/debugging.md b/docs/my-website/docs/proxy/debugging.md index fbcac24a4d6..d9cffbb0cb1 100644 --- a/docs/my-website/docs/proxy/debugging.md +++ b/docs/my-website/docs/proxy/debugging.md @@ -48,7 +48,7 @@ POST Request Sent from LiteLLM: curl -X POST \ https://api.openai.com/v1/chat/completions \ -H 'content-type: application/json' -H 'Authorization: Bearer sk-qnWGUIW9****************************************' \ --d '{"model": "gpt-3.5-turbo", "messages": [{"role": "user", "content": "this is a test request, write a short poem"}]}' +-d '{"model": "gpt-4o", "messages": [{"role": "user", "content": "this is a test request, write a short poem"}]}' ``` ## Debug single request @@ -81,7 +81,7 @@ https://exampleopenaiendpoint-production.up.railway.app/chat/completions \ 20:14:06 - LiteLLM:WARNING: litellm_logging.py:1015 - RAW RESPONSE: -{"id":"chatcmpl-817fc08f0d6c451485d571dab39b26a1","object":"chat.completion","created":1677652288,"model":"gpt-3.5-turbo-0301","system_fingerprint":"fp_44709d6fcb","choices":[{"index":0,"message":{"role":"assistant","content":"\n\nHello there, how may I assist you today?"},"logprobs":null,"finish_reason":"stop"}],"usage":{"prompt_tokens":9,"completion_tokens":12,"total_tokens":21}} +{"id":"chatcmpl-817fc08f0d6c451485d571dab39b26a1","object":"chat.completion","created":1677652288,"model":"gpt-4o-0301","system_fingerprint":"fp_44709d6fcb","choices":[{"index":0,"message":{"role":"assistant","content":"\n\nHello there, how may I assist you today?"},"logprobs":null,"finish_reason":"stop"}],"usage":{"prompt_tokens":9,"completion_tokens":12,"total_tokens":21}} INFO: 127.0.0.1:56155 - "POST /chat/completions HTTP/1.1" 200 OK diff --git a/docs/my-website/docs/proxy/rules.md b/docs/my-website/docs/proxy/rules.md index 60e990d91b4..753547324aa 100644 --- a/docs/my-website/docs/proxy/rules.md +++ b/docs/my-website/docs/proxy/rules.md @@ -34,7 +34,7 @@ curl --location 'http://0.0.0.0:4000/v1/chat/completions' \ --header 'Content-Type: application/json' \ --header 'Authorization: Bearer sk-1234' \ --data '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [{"role":"user","content":"What llm are you?"}], "temperature": 0.7, "max_tokens": 10,