diff --git a/docs/my-website/docs/budget_manager.md b/docs/my-website/docs/budget_manager.md index 6bea96ef9ce..bbbb840c24a 100644 --- a/docs/my-website/docs/budget_manager.md +++ b/docs/my-website/docs/budget_manager.md @@ -51,7 +51,7 @@ if not budget_manager.is_valid_user(user): # check if a given call can be made if budget_manager.get_current_cost(user=user) <= budget_manager.get_total_budget(user): - response = completion(model="gpt-3.5-turbo", messages=[{"role": "user", "content": "Hey, how's it going?"}]) + response = completion(model="gpt-4o", messages=[{"role": "user", "content": "Hey, how's it going?"}]) budget_manager.update_cost(completion_obj=response, user=user) else: response = "Sorry - no budget!" @@ -72,7 +72,7 @@ budget_manager.create_budget(total_budget=10, user=user, duration="daily") input_text = "hello world" output_text = "it's a sunny day in san francisco" -model = "gpt-3.5-turbo" +model = "gpt-4o" budget_manager.update_cost(user=user, model=model, input_text=input_text, output_text=output_text) # 👈 print(budget_manager.get_current_cost(user)) @@ -108,7 +108,7 @@ if not budget_manager.is_valid_user(user): # check if a given call can be made if budget_manager.get_current_cost(user=user) <= budget_manager.get_total_budget(user): - response = completion(model="gpt-3.5-turbo", messages=[{"role": "user", "content": "Hey, how's it going?"}]) + response = completion(model="gpt-4o", messages=[{"role": "user", "content": "Hey, how's it going?"}]) budget_manager.update_cost(completion_obj=response, user=user) else: response = "Sorry - no budget!" @@ -138,7 +138,7 @@ if not budget_manager.is_valid_user(user): # check if a given call can be made if budget_manager.get_current_cost(user=user) <= budget_manager.get_total_budget(user): - response = completion(model="gpt-3.5-turbo", messages=[{"role": "user", "content": "Hey, how's it going?"}]) + response = completion(model="gpt-4o", messages=[{"role": "user", "content": "Hey, how's it going?"}]) budget_manager.update_cost(completion_obj=response, user=user) else: response = "Sorry - no budget!" diff --git a/docs/my-website/docs/exception_mapping.md b/docs/my-website/docs/exception_mapping.md index efdada2a1eb..ac7ae478d20 100644 --- a/docs/my-website/docs/exception_mapping.md +++ b/docs/my-website/docs/exception_mapping.md @@ -69,7 +69,7 @@ except openai.APITimeoutError as e: import litellm try: response = litellm.completion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[ { "role": "user", diff --git a/docs/my-website/docs/load_test_rpm.md b/docs/my-website/docs/load_test_rpm.md index b7621a76468..36095651773 100644 --- a/docs/my-website/docs/load_test_rpm.md +++ b/docs/my-website/docs/load_test_rpm.md @@ -36,7 +36,7 @@ model_list = [ { "model_name": "fake-openai-endpoint", "litellm_params": { - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "api_key": "my-fake-key", "api_base": "http://0.0.0.0:8080", "rpm": 100 @@ -45,7 +45,7 @@ model_list = [ { "model_name": "fake-openai-endpoint", "litellm_params": { - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "api_key": "my-fake-key", "api_base": "http://0.0.0.0:8081", "rpm": 100 diff --git a/docs/my-website/docs/mcp_guardrail.md b/docs/my-website/docs/mcp_guardrail.md index c1f2fbec044..3fdedfe8576 100644 --- a/docs/my-website/docs/mcp_guardrail.md +++ b/docs/my-website/docs/mcp_guardrail.md @@ -42,7 +42,7 @@ curl http://localhost:4000/chat/completions \ -H "Content-Type: application/json" \ -H "Authorization: Bearer sk-1234" \ -d '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ {"role": "user", "content": "My credit card is 4111-1111-1111-1111 and my email is john@example.com"} ], @@ -68,7 +68,7 @@ client = openai.OpenAI( # This request will trigger MCP guardrails response = client.chat.completions.create( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[ {"role": "user", "content": "Send an email to 555-123-4567 with my SSN 123-45-6789"} ], diff --git a/docs/my-website/docs/migration.md b/docs/my-website/docs/migration.md index e1af07d4684..b8b98e4667a 100644 --- a/docs/my-website/docs/migration.md +++ b/docs/my-website/docs/migration.md @@ -18,12 +18,12 @@ When we have breaking changes (i.e. going from 1.x.x to 2.x.x), we will document - *NEW* default exception - `APIConnectionError` (prev. `APIError`) - litellm.get_max_tokens() now returns an int not a dict ```python - max_tokens = litellm.get_max_tokens("gpt-3.5-turbo") # returns an int not a dict + max_tokens = litellm.get_max_tokens("gpt-4o") # returns an int not a dict assert max_tokens==4097 ``` - Streaming - OpenAI Chunks now return `None` for empty stream chunks. This is how to process stream chunks with content ```python - response = litellm.completion(model="gpt-3.5-turbo", messages=messages, stream=True) + response = litellm.completion(model="gpt-4o", messages=messages, stream=True) for part in response: print(part.choices[0].delta.content or "") ``` diff --git a/docs/my-website/docs/old_guardrails.md b/docs/my-website/docs/old_guardrails.md index 73448666c43..0ce9c7395eb 100644 --- a/docs/my-website/docs/old_guardrails.md +++ b/docs/my-website/docs/old_guardrails.md @@ -11,9 +11,9 @@ Setup Prompt Injection Detection, Secret Detection on LiteLLM Proxy ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: - model: openai/gpt-3.5-turbo + model: openai/gpt-4o api_key: sk-xxxxxxx litellm_settings: @@ -56,7 +56,7 @@ curl --location 'http://localhost:4000/chat/completions' \ --header 'Authorization: Bearer sk-1234' \ --header 'Content-Type: application/json' \ --data '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", @@ -280,7 +280,7 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \ --header 'Content-Type: application/json' \ --header 'Authorization: Bearer $LITELLM_VIRTUAL_KEY' \ --data '{ -"model": "gpt-3.5-turbo", +"model": "gpt-4o", "messages": [ { "role": "user", diff --git a/docs/my-website/docs/proxy_api.md b/docs/my-website/docs/proxy_api.md index 7612645fb54..31f33a9d72c 100644 --- a/docs/my-website/docs/proxy_api.md +++ b/docs/my-website/docs/proxy_api.md @@ -15,7 +15,7 @@ os.environ["COHERE_API_KEY"] = "your-api-key" messages = [{ "content": "Hello, how are you?","role": "user"}] # openai call -response = completion(model="gpt-3.5-turbo", messages=messages) +response = completion(model="gpt-4o", messages=messages) # cohere call response = completion("command-nightly", messages) @@ -31,8 +31,8 @@ For a complete list of models/providers that you can call with LiteLLM, [check o * OpenAI models - [OpenAI docs](./providers/openai.md) * gpt-4 - * gpt-3.5-turbo - * gpt-3.5-turbo-16k + * gpt-4o + * gpt-4o-16k * Llama2 models - [TogetherAI docs](./providers/togetherai.md) * togethercomputer/llama-2-70b-chat * togethercomputer/llama-2-70b diff --git a/docs/my-website/docs/reasoning_content.md b/docs/my-website/docs/reasoning_content.md index 8bf59f66a33..9b0ca6e7422 100644 --- a/docs/my-website/docs/reasoning_content.md +++ b/docs/my-website/docs/reasoning_content.md @@ -520,7 +520,7 @@ assert litellm.supports_reasoning(model="anthropic/claude-3-7-sonnet-20250219") assert litellm.supports_reasoning(model="deepseek/deepseek-chat") == True # Example models that do not support reasoning -assert litellm.supports_reasoning(model="openai/gpt-3.5-turbo") == False +assert litellm.supports_reasoning(model="openai/gpt-4o") == False ``` diff --git a/docs/my-website/docs/routing.md b/docs/my-website/docs/routing.md index 67e7f681147..cd4ef9d335b 100644 --- a/docs/my-website/docs/routing.md +++ b/docs/my-website/docs/routing.md @@ -34,7 +34,7 @@ Loadbalance across multiple [azure](./providers/azure)/[bedrock](./providers/bed from litellm import Router model_list = [{ # list of model deployments - "model_name": "gpt-3.5-turbo", # model alias -> loadbalance between models with same `model_name` + "model_name": "gpt-4o", # model alias -> loadbalance between models with same `model_name` "litellm_params": { # params for litellm completion/embedding call "model": "azure/chatgpt-v-2", # actual model name "api_key": os.getenv("AZURE_API_KEY"), @@ -42,7 +42,7 @@ model_list = [{ # list of model deployments "api_base": os.getenv("AZURE_API_BASE") } }, { - "model_name": "gpt-3.5-turbo", + "model_name": "gpt-4o", "litellm_params": { # params for litellm completion/embedding call "model": "azure/chatgpt-functioncalling", "api_key": os.getenv("AZURE_API_KEY"), @@ -50,9 +50,9 @@ model_list = [{ # list of model deployments "api_base": os.getenv("AZURE_API_BASE") } }, { - "model_name": "gpt-3.5-turbo", + "model_name": "gpt-4o", "litellm_params": { # params for litellm completion/embedding call - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "api_key": os.getenv("OPENAI_API_KEY"), } }, { @@ -76,8 +76,8 @@ model_list = [{ # list of model deployments router = Router(model_list=model_list) # openai.ChatCompletion.create replacement -# requests with model="gpt-3.5-turbo" will pick a deployment where model_name="gpt-3.5-turbo" -response = await router.acompletion(model="gpt-3.5-turbo", +# requests with model="gpt-4o" will pick a deployment where model_name="gpt-4o" +response = await router.acompletion(model="gpt-4o", messages=[{"role": "user", "content": "Hey, how's it going?"}]) print(response) @@ -101,17 +101,17 @@ See detailed proxy loadbalancing/fallback docs [here](./proxy/reliability.md) 1. Setup model_list with multiple deployments ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: model: azure/ api_base: api_key: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: model: azure/gpt-turbo-small-ca api_base: https://my-endpoint-canada-berri992.openai.azure.com/ api_key: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: model: azure/gpt-turbo-large api_base: https://openai-france-1234.openai.azure.com/ @@ -131,7 +131,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \ -H 'Content-Type: application/json' \ -H 'Authorization: Bearer sk-1234' \ -d '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ {"role": "user", "content": "Hi there!"} ], @@ -174,14 +174,14 @@ You can also set a `weight` param, to specify which model should get picked when ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: model: azure/chatgpt-v-2 api_key: os.environ/AZURE_API_KEY api_version: os.environ/AZURE_API_VERSION api_base: os.environ/AZURE_API_BASE rpm: 900 - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: model: azure/chatgpt-functioncalling api_key: os.environ/AZURE_API_KEY @@ -197,7 +197,7 @@ from litellm import Router import asyncio model_list = [{ # list of model deployments - "model_name": "gpt-3.5-turbo", # model alias + "model_name": "gpt-4o", # model alias "litellm_params": { # params for litellm completion/embedding call "model": "azure/chatgpt-v-2", # actual model name "api_key": os.getenv("AZURE_API_KEY"), @@ -206,7 +206,7 @@ model_list = [{ # list of model deployments "rpm": 900, # requests per minute for this API } }, { - "model_name": "gpt-3.5-turbo", + "model_name": "gpt-4o", "litellm_params": { # params for litellm completion/embedding call "model": "azure/chatgpt-functioncalling", "api_key": os.getenv("AZURE_API_KEY"), @@ -220,7 +220,7 @@ model_list = [{ # list of model deployments router = Router(model_list=model_list, routing_strategy="simple-shuffle") async def router_acompletion(): response = await router.acompletion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{"role": "user", "content": "Hey, how's it going?"}] ) print(response) @@ -236,14 +236,14 @@ asyncio.run(router_acompletion()) ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: model: azure/chatgpt-v-2 api_key: os.environ/AZURE_API_KEY api_version: os.environ/AZURE_API_VERSION api_base: os.environ/AZURE_API_BASE weight: 9 - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: model: azure/chatgpt-functioncalling api_key: os.environ/AZURE_API_KEY @@ -259,7 +259,7 @@ from litellm import Router import asyncio model_list = [{ - "model_name": "gpt-3.5-turbo", # model alias + "model_name": "gpt-4o", # model alias "litellm_params": { "model": "azure/chatgpt-v-2", # actual model name "api_key": os.getenv("AZURE_API_KEY"), @@ -268,7 +268,7 @@ model_list = [{ "weight": 9, # pick this 90% of the time } }, { - "model_name": "gpt-3.5-turbo", + "model_name": "gpt-4o", "litellm_params": { "model": "azure/chatgpt-functioncalling", "api_key": os.getenv("AZURE_API_KEY"), @@ -282,7 +282,7 @@ model_list = [{ router = Router(model_list=model_list, routing_strategy="simple-shuffle") async def router_acompletion(): response = await router.acompletion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{"role": "user", "content": "Hey, how's it going?"}] ) print(response) @@ -319,7 +319,7 @@ from litellm import Router model_list = [{ # list of model deployments - "model_name": "gpt-3.5-turbo", # model alias + "model_name": "gpt-4o", # model alias "litellm_params": { # params for litellm completion/embedding call "model": "azure/chatgpt-v-2", # actual model name "api_key": os.getenv("AZURE_API_KEY"), @@ -329,7 +329,7 @@ model_list = [{ # list of model deployments "rpm": 10000, }, }, { - "model_name": "gpt-3.5-turbo", + "model_name": "gpt-4o", "litellm_params": { # params for litellm completion/embedding call "model": "azure/chatgpt-functioncalling", "api_key": os.getenv("AZURE_API_KEY"), @@ -339,9 +339,9 @@ model_list = [{ # list of model deployments "rpm": 1000, }, }, { - "model_name": "gpt-3.5-turbo", + "model_name": "gpt-4o", "litellm_params": { # params for litellm completion/embedding call - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "api_key": os.getenv("OPENAI_API_KEY"), "tpm": 100000, "rpm": 1000, @@ -355,7 +355,7 @@ router = Router(model_list=model_list, enable_pre_call_checks=True, # enables router rate limits for concurrent calls ) -response = await router.acompletion(model="gpt-3.5-turbo", +response = await router.acompletion(model="gpt-4o", messages=[{"role": "user", "content": "Hey, how's it going?"}] print(response) @@ -367,7 +367,7 @@ print(response) ```yaml model_list: - - model_name: gpt-3.5-turbo # model alias + - model_name: gpt-4o # model alias litellm_params: # params for litellm completion/embedding call model: azure/chatgpt-v-2 # actual model name api_key: os.environ/AZURE_API_KEY @@ -375,9 +375,9 @@ model_list: api_base: os.environ/AZURE_API_BASE tpm: 100000 rpm: 10000 - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: # params for litellm completion/embedding call - model: gpt-3.5-turbo + model: gpt-4o api_key: os.getenv(OPENAI_API_KEY) tpm: 100000 rpm: 1000 @@ -406,7 +406,7 @@ curl --location 'http://localhost:4000/v1/chat/completions' \ --header 'Content-Type: application/json' \ --header 'Authorization: Bearer sk-1234' \ --data '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [{"role": "user", "content": "Hey, how's it going?"}] }' ``` @@ -524,7 +524,7 @@ from litellm import Router model_list = [{ # list of model deployments - "model_name": "gpt-3.5-turbo", # model alias + "model_name": "gpt-4o", # model alias "litellm_params": { # params for litellm completion/embedding call "model": "azure/chatgpt-v-2", # actual model name "api_key": os.getenv("AZURE_API_KEY"), @@ -534,7 +534,7 @@ model_list = [{ # list of model deployments "tpm": 100000, "rpm": 10000, }, { - "model_name": "gpt-3.5-turbo", + "model_name": "gpt-4o", "litellm_params": { # params for litellm completion/embedding call "model": "azure/chatgpt-functioncalling", "api_key": os.getenv("AZURE_API_KEY"), @@ -544,9 +544,9 @@ model_list = [{ # list of model deployments "tpm": 100000, "rpm": 1000, }, { - "model_name": "gpt-3.5-turbo", + "model_name": "gpt-4o", "litellm_params": { # params for litellm completion/embedding call - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "api_key": os.getenv("OPENAI_API_KEY"), }, "tpm": 100000, @@ -560,7 +560,7 @@ router = Router(model_list=model_list, enable_pre_call_check=True, # enables router rate limits for concurrent calls ) -response = await router.acompletion(model="gpt-3.5-turbo", +response = await router.acompletion(model="gpt-4o", messages=[{"role": "user", "content": "Hey, how's it going?"}] print(response) @@ -580,7 +580,7 @@ from litellm import Router import asyncio model_list = [{ # list of model deployments - "model_name": "gpt-3.5-turbo", # model alias + "model_name": "gpt-4o", # model alias "litellm_params": { # params for litellm completion/embedding call "model": "azure/chatgpt-v-2", # actual model name "api_key": os.getenv("AZURE_API_KEY"), @@ -588,7 +588,7 @@ model_list = [{ # list of model deployments "api_base": os.getenv("AZURE_API_BASE"), } }, { - "model_name": "gpt-3.5-turbo", + "model_name": "gpt-4o", "litellm_params": { # params for litellm completion/embedding call "model": "azure/chatgpt-functioncalling", "api_key": os.getenv("AZURE_API_KEY"), @@ -596,9 +596,9 @@ model_list = [{ # list of model deployments "api_base": os.getenv("AZURE_API_BASE"), } }, { - "model_name": "gpt-3.5-turbo", + "model_name": "gpt-4o", "litellm_params": { # params for litellm completion/embedding call - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "api_key": os.getenv("OPENAI_API_KEY"), } }] @@ -607,7 +607,7 @@ model_list = [{ # list of model deployments router = Router(model_list=model_list, routing_strategy="least-busy") async def router_acompletion(): response = await router.acompletion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{"role": "user", "content": "Hey, how's it going?"}] ) print(response) @@ -750,12 +750,12 @@ import asyncio model_list = [ { - "model_name": "gpt-3.5-turbo", + "model_name": "gpt-4o", "litellm_params": {"model": "gpt-4"}, "model_info": {"id": "openai-gpt-4"}, }, { - "model_name": "gpt-3.5-turbo", + "model_name": "gpt-4o", "litellm_params": {"model": "groq/llama3-8b-8192"}, "model_info": {"id": "groq-llama"}, }, @@ -765,7 +765,7 @@ model_list = [ router = Router(model_list=model_list, routing_strategy="cost-based-routing") async def router_acompletion(): response = await router.acompletion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{"role": "user", "content": "Hey, how's it going?"}] ) print(response) @@ -785,7 +785,7 @@ Set `litellm_params["input_cost_per_token"]` and `litellm_params["output_cost_pe ```python model_list = [ { - "model_name": "gpt-3.5-turbo", + "model_name": "gpt-4o", "litellm_params": { "model": "azure/chatgpt-v-2", "input_cost_per_token": 0.00003, @@ -794,7 +794,7 @@ model_list = [ "model_info": {"id": "chatgpt-v-experimental"}, }, { - "model_name": "gpt-3.5-turbo", + "model_name": "gpt-4o", "litellm_params": { "model": "azure/chatgpt-v-1", "input_cost_per_token": 0.000000001, @@ -803,7 +803,7 @@ model_list = [ "model_info": {"id": "chatgpt-v-1"}, }, { - "model_name": "gpt-3.5-turbo", + "model_name": "gpt-4o", "litellm_params": { "model": "azure/chatgpt-v-5", "input_cost_per_token": 10, @@ -816,7 +816,7 @@ model_list = [ router = Router(model_list=model_list, routing_strategy="cost-based-routing") async def router_acompletion(): response = await router.acompletion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{"role": "user", "content": "Hey, how's it going?"}] ) print(response) @@ -932,7 +932,7 @@ model_list = [ router = Router(model_list=model_list, routing_strategy="cost-based-routing") response = await router.acompletion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{"role": "user", "content": "Hey, how's it going?"}] ) print(response) @@ -1007,7 +1007,7 @@ user_message = "Hello, whats the weather in San Francisco??" messages = [{"content": user_message, "role": "user"}] # normal call -response = router.completion(model="gpt-3.5-turbo", messages=messages) +response = router.completion(model="gpt-4o", messages=messages) print(f"response: {response}") ``` @@ -1190,7 +1190,7 @@ user_message = "Hello, whats the weather in San Francisco??" messages = [{"content": user_message, "role": "user"}] # normal call -response = router.completion(model="gpt-3.5-turbo", messages=messages) +response = router.completion(model="gpt-4o", messages=messages) print(f"response: {response}") ``` @@ -1209,7 +1209,7 @@ user_message = "Hello, whats the weather in San Francisco??" messages = [{"content": user_message, "role": "user"}] # normal call -response = router.completion(model="gpt-3.5-turbo", messages=messages) +response = router.completion(model="gpt-4o", messages=messages) print(f"response: {response}") ``` @@ -1260,7 +1260,7 @@ allowed_fails_policy = AllowedFailsPolicy( router = litellm.Router( model_list=[ { - "model_name": "gpt-3.5-turbo", # openai model name + "model_name": "gpt-4o", # openai model name "litellm_params": { # params for litellm completion/embedding call "model": "azure/chatgpt-v-2", "api_key": os.getenv("AZURE_API_KEY"), @@ -1374,7 +1374,7 @@ For 'eu-region' filtering, Set 'region_name' of deployment. ```python model_list = [ { - "model_name": "gpt-3.5-turbo", # model group name + "model_name": "gpt-4o", # model group name "litellm_params": { # params for litellm completion/embedding call "model": "azure/chatgpt-v-2", "api_key": os.getenv("AZURE_API_KEY"), @@ -1385,9 +1385,9 @@ model_list = [ }, }, { - "model_name": "gpt-3.5-turbo", # model group name + "model_name": "gpt-4o", # model group name "litellm_params": { # params for litellm completion/embedding call - "model": "gpt-3.5-turbo-1106", + "model": "gpt-4o-1106", "api_key": os.getenv("OPENAI_API_KEY"), }, }, @@ -1413,7 +1413,7 @@ router = Router(model_list=model_list, enable_pre_call_checks=True) ```python """ -- Give a gpt-3.5-turbo model group with different context windows (4k vs. 16k) +- Give a gpt-4o model group with different context windows (4k vs. 16k) - Send a 5k prompt - Assert it works """ @@ -1422,7 +1422,7 @@ import os model_list = [ { - "model_name": "gpt-3.5-turbo", # model group name + "model_name": "gpt-4o", # model group name "litellm_params": { # params for litellm completion/embedding call "model": "azure/chatgpt-v-2", "api_key": os.getenv("AZURE_API_KEY"), @@ -1435,9 +1435,9 @@ model_list = [ } }, { - "model_name": "gpt-3.5-turbo", # model group name + "model_name": "gpt-4o", # model group name "litellm_params": { # params for litellm completion/embedding call - "model": "gpt-3.5-turbo-1106", + "model": "gpt-4o-1106", "api_key": os.getenv("OPENAI_API_KEY"), }, }, @@ -1448,7 +1448,7 @@ router = Router(model_list=model_list, enable_pre_call_checks=True) text = "What is the meaning of 42?" * 5000 response = router.completion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[ {"role": "system", "content": text}, {"role": "user", "content": "Who was Alexander?"}, @@ -1462,7 +1462,7 @@ print(f"response: {response}") ```python """ -- Give 2 gpt-3.5-turbo deployments, in eu + non-eu regions +- Give 2 gpt-4o deployments, in eu + non-eu regions - Make a call - Assert it picks the eu-region model """ @@ -1472,7 +1472,7 @@ import os model_list = [ { - "model_name": "gpt-3.5-turbo", # model group name + "model_name": "gpt-4o", # model group name "litellm_params": { # params for litellm completion/embedding call "model": "azure/chatgpt-v-2", "api_key": os.getenv("AZURE_API_KEY"), @@ -1485,9 +1485,9 @@ model_list = [ } }, { - "model_name": "gpt-3.5-turbo", # model group name + "model_name": "gpt-4o", # model group name "litellm_params": { # params for litellm completion/embedding call - "model": "gpt-3.5-turbo-1106", + "model": "gpt-4o-1106", "api_key": os.getenv("OPENAI_API_KEY"), }, "model_info": { @@ -1499,7 +1499,7 @@ model_list = [ router = Router(model_list=model_list, enable_pre_call_checks=True) response = router.completion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{"role": "user", "content": "Who was Alexander?"}], ) @@ -1539,14 +1539,14 @@ async def test_acompletion_caching_on_router_caching_groups(): litellm.set_verbose = True model_list = [ { - "model_name": "openai-gpt-3.5-turbo", + "model_name": "openai-gpt-4o", "litellm_params": { - "model": "gpt-3.5-turbo-0613", + "model": "gpt-4o-0613", "api_key": os.getenv("OPENAI_API_KEY"), }, }, { - "model_name": "azure-gpt-3.5-turbo", + "model_name": "azure-gpt-4o", "litellm_params": { "model": "azure/chatgpt-v-2", "api_key": os.getenv("AZURE_API_KEY"), @@ -1562,11 +1562,11 @@ async def test_acompletion_caching_on_router_caching_groups(): start_time = time.time() router = Router(model_list=model_list, cache_responses=True, - caching_groups=[("openai-gpt-3.5-turbo", "azure-gpt-3.5-turbo")]) - response1 = await router.acompletion(model="openai-gpt-3.5-turbo", messages=messages, temperature=1) + caching_groups=[("openai-gpt-4o", "azure-gpt-4o")]) + response1 = await router.acompletion(model="openai-gpt-4o", messages=messages, temperature=1) print(f"response1: {response1}") await asyncio.sleep(1) # add cache is async, async sleep for cache to get set - response2 = await router.acompletion(model="azure-gpt-3.5-turbo", messages=messages, temperature=1) + response2 = await router.acompletion(model="azure-gpt-4o", messages=messages, temperature=1) assert response1.id == response2.id assert len(response1.choices[0].message.content) > 0 assert response1.choices[0].message.content == response2.choices[0].message.content @@ -1597,9 +1597,9 @@ import asyncio router = Router( model_list=[ { - "model_name": "gpt-3.5-turbo", + "model_name": "gpt-4o", "litellm_params": { - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "api_key": "bad_key", }, } @@ -1616,7 +1616,7 @@ async def main(): try: await router.acompletion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{"role": "user", "content": "Hey, how's it going?"}], ) except Exception as e: @@ -1701,7 +1701,7 @@ You can also set default params for litellm completion/embedding calls. Here's h ```python from litellm import Router -fallback_dict = {"gpt-3.5-turbo": "gpt-3.5-turbo-16k"} +fallback_dict = {"gpt-4o": "gpt-4o-16k"} router = Router(model_list=model_list, default_litellm_params={"context_window_fallback_dict": fallback_dict}) @@ -1710,7 +1710,7 @@ user_message = "Hello, whats the weather in San Francisco??" messages = [{"content": user_message, "role": "user"}] # normal call -response = router.completion(model="gpt-3.5-turbo", messages=messages) +response = router.completion(model="gpt-4o", messages=messages) print(f"response: {response}") ``` @@ -1754,7 +1754,7 @@ router = Router(model_list=model_list, routing_strategy="simple-shuffle") # router completion call response = router.completion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{ "role": "user", "content": "Hi who are you"}] ) ``` diff --git a/docs/my-website/docs/rules.md b/docs/my-website/docs/rules.md index 97da9096db4..98eef62f6dd 100644 --- a/docs/my-website/docs/rules.md +++ b/docs/my-website/docs/rules.md @@ -18,7 +18,7 @@ def my_custom_rule(input): # receives the model response litellm.post_call_rules = [my_custom_rule] # have these be functions that can be called to fail a call -response = litellm.completion(model="gpt-3.5-turbo", messages=[{"role": "user", +response = litellm.completion(model="gpt-4o", messages=[{"role": "user", "content": "Hey, how's it going?"}], fallbacks=["openrouter/gryphe/mythomax-l2-13b"]) ``` @@ -63,7 +63,7 @@ def my_custom_rule(input): # receives the model response litellm.pre_call_rules = [my_custom_rule] # have these be functions that can be called to fail a call -response = litellm.completion(model="gpt-3.5-turbo", messages=[{"role": "user", "content": "Hey, how's it going?"}]) +response = litellm.completion(model="gpt-4o", messages=[{"role": "user", "content": "Hey, how's it going?"}]) ``` ### Example 2: Fallback to uncensored model if llm refuses to answer @@ -84,6 +84,6 @@ def my_custom_rule(input): # receives the model response litellm.post_call_rules = [my_custom_rule] # have these be functions that can be called to fail a call -response = litellm.completion(model="gpt-3.5-turbo", messages=[{"role": "user", +response = litellm.completion(model="gpt-4o", messages=[{"role": "user", "content": "Hey, how's it going?"}], fallbacks=["openrouter/gryphe/mythomax-l2-13b"]) ``` \ No newline at end of file diff --git a/docs/my-website/docs/scheduler.md b/docs/my-website/docs/scheduler.md index 9b84c374e3b..e8a80240d28 100644 --- a/docs/my-website/docs/scheduler.md +++ b/docs/my-website/docs/scheduler.md @@ -32,9 +32,9 @@ from litellm import Router router = Router( model_list=[ { - "model_name": "gpt-3.5-turbo", + "model_name": "gpt-4o", "litellm_params": { - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "mock_response": "Hello world this is Macintosh!", # fakes the LLM API call "rpm": 1, }, @@ -47,7 +47,7 @@ router = Router( try: _response = await router.acompletion( # 👈 ADDS TO QUEUE + POLLS + MAKES CALL - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{"role": "user", "content": "Hey!"}], priority=0, # 👈 LOWER IS BETTER ) @@ -67,7 +67,7 @@ curl -X POST 'http://localhost:4000/chat/completions' \ -H 'Content-Type: application/json' \ -H 'Authorization: Bearer sk-1234' \ -D '{ - "model": "gpt-3.5-turbo-fake-model", + "model": "gpt-4o-fake-model", "messages": [ { "role": "user", @@ -89,7 +89,7 @@ client = openai.OpenAI( # request sent to model set on litellm proxy, `litellm --model` response = client.chat.completions.create( - model="gpt-3.5-turbo", + model="gpt-4o", messages = [ { "role": "user", @@ -118,9 +118,9 @@ from litellm import Router router = Router( model_list=[ { - "model_name": "gpt-3.5-turbo", + "model_name": "gpt-4o", "litellm_params": { - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "mock_response": "Hello world this is Macintosh!", # fakes the LLM API call "rpm": 1, }, @@ -134,7 +134,7 @@ router = Router( try: _response = await router.acompletion( # 👈 ADDS TO QUEUE + POLLS + MAKES CALL - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{"role": "user", "content": "Hey!"}], priority=0, # 👈 LOWER IS BETTER ) @@ -146,9 +146,9 @@ except Exception as e: ```yaml model_list: - - model_name: gpt-3.5-turbo-fake-model + - model_name: gpt-4o-fake-model litellm_params: - model: gpt-3.5-turbo + model: gpt-4o mock_response: "hello world!" api_key: my-good-key @@ -172,7 +172,7 @@ curl -X POST 'http://localhost:4000/queue/chat/completions' \ -H 'Content-Type: application/json' \ -H 'Authorization: Bearer sk-1234' \ -D '{ - "model": "gpt-3.5-turbo-fake-model", + "model": "gpt-4o-fake-model", "messages": [ { "role": "user", diff --git a/docs/my-website/docs/text_completion.md b/docs/my-website/docs/text_completion.md index 234494c2dd9..e3273499899 100644 --- a/docs/my-website/docs/text_completion.md +++ b/docs/my-website/docs/text_completion.md @@ -24,7 +24,7 @@ import TabItem from '@theme/TabItem'; from litellm import text_completion response = text_completion( - model="gpt-3.5-turbo-instruct", + model="gpt-4o-instruct", prompt="Say this is a test", max_tokens=7 ) @@ -37,9 +37,9 @@ response = text_completion( ```yaml model_list: - - model_name: gpt-3.5-turbo-instruct + - model_name: gpt-4o-instruct litellm_params: - model: text-completion-openai/gpt-3.5-turbo-instruct # The `text-completion-openai/` prefix will call openai.completions.create + model: text-completion-openai/gpt-4o-instruct # The `text-completion-openai/` prefix will call openai.completions.create api_key: os.environ/OPENAI_API_KEY - model_name: text-davinci-003 litellm_params: @@ -64,7 +64,7 @@ from openai import OpenAI client = OpenAI(api_key="", base_url="http://0.0.0.0:4000") response = client.completions.create( - model="gpt-3.5-turbo-instruct", + model="gpt-4o-instruct", prompt="Say this is a test", max_tokens=7 ) @@ -80,7 +80,7 @@ curl --location 'http://0.0.0.0:4000/completions' \ --header 'Content-Type: application/json' \ --header 'Authorization: Bearer sk-1234' \ --data '{ - "model": "gpt-3.5-turbo-instruct", + "model": "gpt-4o-instruct", "prompt": "Say this is a test", "max_tokens": 7 }' @@ -133,7 +133,7 @@ Here's the exact JSON output format you can expect from completion calls: "id": "cmpl-uqkvlQyYK7bGYrRHQ0eXlWi7", "object": "text_completion", "created": 1589478378, - "model": "gpt-3.5-turbo-instruct", + "model": "gpt-4o-instruct", "system_fingerprint": "fp_44709d6fcb", "choices": [ { @@ -167,7 +167,7 @@ Here's the exact JSON output format you can expect from completion calls: "finish_reason": null } ], - "model": "gpt-3.5-turbo-instruct" + "model": "gpt-4o-instruct" "system_fingerprint": "fp_44709d6fcb", } diff --git a/docs/my-website/docs/traffic_mirroring.md b/docs/my-website/docs/traffic_mirroring.md index 3bdcb0f1614..92a619f5e80 100644 --- a/docs/my-website/docs/traffic_mirroring.md +++ b/docs/my-website/docs/traffic_mirroring.md @@ -22,7 +22,7 @@ from litellm import Router model_list = [ { - "model_name": "gpt-3.5-turbo", + "model_name": "gpt-4o", "litellm_params": { "model": "azure/chatgpt-v-2", "api_key": "...", @@ -40,9 +40,9 @@ model_list = [ router = Router(model_list=model_list) -# The request to "gpt-3.5-turbo" will trigger a background call to "gpt-4" +# The request to "gpt-4o" will trigger a background call to "gpt-4" response = await router.acompletion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{"role": "user", "content": "How does traffic mirroring work?"}] ) ```