diff --git a/docs/my-website/docs/adding_provider/adding_guardrail_support.md b/docs/my-website/docs/adding_provider/adding_guardrail_support.md index 2646b626ab5..ad76785a33a 100644 --- a/docs/my-website/docs/adding_provider/adding_guardrail_support.md +++ b/docs/my-website/docs/adding_provider/adding_guardrail_support.md @@ -321,7 +321,7 @@ curl -X POST 'http://localhost:4000/{my_endpoint}' \ -H 'Content-Type: application/json' \ -H 'Authorization: Bearer your-api-key' \ -d '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [{"role": "user", "content": "Hello"}], "guardrails": ["test"] }' diff --git a/docs/my-website/docs/adding_provider/generic_prompt_management_api.md b/docs/my-website/docs/adding_provider/generic_prompt_management_api.md index d1b119d94c5..c6896ac1243 100644 --- a/docs/my-website/docs/adding_provider/generic_prompt_management_api.md +++ b/docs/my-website/docs/adding_provider/generic_prompt_management_api.md @@ -131,9 +131,9 @@ Add to `config.yaml`: ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: - model: openai/gpt-3.5-turbo + model: openai/gpt-4o api_key: os.environ/OPENAI_API_KEY prompts: diff --git a/docs/my-website/docs/caching/all_caches.md b/docs/my-website/docs/caching/all_caches.md index 6f81da9105a..02dda99f7a3 100644 --- a/docs/my-website/docs/caching/all_caches.md +++ b/docs/my-website/docs/caching/all_caches.md @@ -39,11 +39,11 @@ litellm.cache = Cache(type="redis", host=, port=, password=', messages)` | | gpt-4-1106-preview | `completion('azure/', messages)` | | gpt-4-0125-preview | `completion('azure/', messages)` | -| gpt-3.5-turbo | `completion('azure/', messages)` | -| gpt-3.5-turbo-0301 | `completion('azure/', messages)` | -| gpt-3.5-turbo-0613 | `completion('azure/', messages)` | -| gpt-3.5-turbo-16k | `completion('azure/', messages)` | -| gpt-3.5-turbo-16k-0613 | `completion('azure/', messages)` +| gpt-4o | `completion('azure/', messages)` | +| gpt-4o-0301 | `completion('azure/', messages)` | +| gpt-4o-0613 | `completion('azure/', messages)` | +| gpt-4o-16k | `completion('azure/', messages)` | +| gpt-4o-16k-0613 | `completion('azure/', messages)` ## Azure OpenAI Vision Models | Model Name | Function Call | @@ -524,8 +524,8 @@ Use `model="azure_text/"` | Model Name | Function Call | |---------------------|----------------------------------------------------| -| gpt-3.5-turbo-instruct | `response = completion(model="azure_text/", messages=messages)` | -| gpt-3.5-turbo-instruct-0914 | `response = completion(model="azure_text/", messages=messages)` | +| gpt-4o-instruct | `response = completion(model="azure_text/", messages=messages)` | +| gpt-4o-instruct-0914 | `response = completion(model="azure_text/", messages=messages)` | ```python @@ -604,7 +604,7 @@ response = litellm.completion( ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: model: azure/chatgpt-v-2 api_base: https://openai-gpt-4-test-v-1.openai.azure.com/ @@ -620,7 +620,7 @@ model_list: Here is an example of setting up `tenant_id`, `client_id`, `client_secret` in your litellm proxy `config.yaml` ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: model: azure/chatgpt-v-2 api_base: https://openai-gpt-4-test-v-1.openai.azure.com/ @@ -637,7 +637,7 @@ Test it curl --location 'http://0.0.0.0:4000/chat/completions' \ --header 'Content-Type: application/json' \ --data ' { - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", @@ -657,7 +657,7 @@ Example video of using `tenant_id`, `client_id`, `client_secret` with LiteLLM Pr Here is an example of setting up `client_id`, `azure_username`, `azure_password` in your litellm proxy `config.yaml` ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: model: azure/chatgpt-v-2 api_base: https://openai-gpt-4-test-v-1.openai.azure.com/ @@ -674,7 +674,7 @@ Test it curl --location 'http://0.0.0.0:4000/chat/completions' \ --header 'Content-Type: application/json' \ --data ' { - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", @@ -746,7 +746,7 @@ export AZURE_CLIENT_SECRET="" ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: model: azure/your-deployment-name api_base: https://openai-gpt-4-test-v-1.openai.azure.com/ @@ -761,7 +761,7 @@ Perfect for AKS clusters, Azure VMs, or other managed environments where Azure a ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: model: azure/your-deployment-name api_base: https://openai-gpt-4-test-v-1.openai.azure.com/ @@ -776,7 +776,7 @@ If you're authenticated via `az login`, no additional configuration needed: ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: model: azure/your-deployment-name api_base: https://openai-gpt-4-test-v-1.openai.azure.com/ @@ -1150,7 +1150,7 @@ pip install litellm from litellm import Router model_list = [{ # list of model deployments - "model_name": "gpt-3.5-turbo", # openai model name + "model_name": "gpt-4o", # openai model name "litellm_params": { # params for litellm completion/embedding call "model": "azure/chatgpt-v-2", "api_key": os.getenv("AZURE_API_KEY"), @@ -1160,7 +1160,7 @@ model_list = [{ # list of model deployments "tpm": 240000, "rpm": 1800 }, { - "model_name": "gpt-3.5-turbo", # openai model name + "model_name": "gpt-4o", # openai model name "litellm_params": { # params for litellm completion/embedding call "model": "azure/chatgpt-functioncalling", "api_key": os.getenv("AZURE_API_KEY"), @@ -1170,9 +1170,9 @@ model_list = [{ # list of model deployments "tpm": 240000, "rpm": 1800 }, { - "model_name": "gpt-3.5-turbo", # openai model name + "model_name": "gpt-4o", # openai model name "litellm_params": { # params for litellm completion/embedding call - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "api_key": os.getenv("OPENAI_API_KEY"), }, "tpm": 1000000, @@ -1182,7 +1182,7 @@ model_list = [{ # list of model deployments router = Router(model_list=model_list) # openai.chat.completions.create replacement -response = router.completion(model="gpt-3.5-turbo", +response = router.completion(model="gpt-4o", messages=[{"role": "user", "content": "Hey, how's it going?"}] print(response) diff --git a/docs/my-website/docs/providers/custom_llm_server.md b/docs/my-website/docs/providers/custom_llm_server.md index 4fcbf8942ce..ac350af7446 100644 --- a/docs/my-website/docs/providers/custom_llm_server.md +++ b/docs/my-website/docs/providers/custom_llm_server.md @@ -31,7 +31,7 @@ from litellm import CustomLLM, completion, get_llm_provider class MyCustomLLM(CustomLLM): def completion(self, *args, **kwargs) -> litellm.ModelResponse: return litellm.completion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{"role": "user", "content": "Hello world"}], mock_response="Hi!", ) # type: ignore @@ -62,14 +62,14 @@ from litellm import CustomLLM, completion, get_llm_provider class MyCustomLLM(CustomLLM): def completion(self, *args, **kwargs) -> litellm.ModelResponse: return litellm.completion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{"role": "user", "content": "Hello world"}], mock_response="Hi!", ) # type: ignore async def acompletion(self, *args, **kwargs) -> litellm.ModelResponse: return litellm.completion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{"role": "user", "content": "Hello world"}], mock_response="Hi!", ) # type: ignore @@ -135,7 +135,7 @@ Expected Response } ], "created": 1721955063, - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "object": "chat.completion", "system_fingerprint": null, "usage": { @@ -356,7 +356,7 @@ from litellm import CustomLLM, completion, get_llm_provider class MyCustomLLM(CustomLLM): async def acompletion(self, *args, **kwargs) -> litellm.ModelResponse: return litellm.completion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{"role": "user", "content": "Hello world"}], mock_response="Hi!", ) # type: ignore @@ -420,7 +420,7 @@ Expected Response "id": "chatcmpl-Bm4qEp4h4vCe7Zi4Gud1MAxTWgibO", "type": "message", "role": "assistant", - "model": "gpt-3.5-turbo-0125", + "model": "gpt-4o-0125", "stop_sequence": null, "usage": { "input_tokens": 18, @@ -455,7 +455,7 @@ class MyCustomLLM(CustomLLM): def completion(self, *args, **kwargs) -> litellm.ModelResponse: assert kwargs["optional_params"] == {"my_custom_param": "my-custom-param"} # 👈 CHECK HERE return litellm.completion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{"role": "user", "content": "Hello world"}], mock_response="Hi!", ) # type: ignore diff --git a/docs/my-website/docs/providers/openrouter.md b/docs/my-website/docs/providers/openrouter.md index 4c79c41cfd5..ee7189f0540 100644 --- a/docs/my-website/docs/providers/openrouter.md +++ b/docs/my-website/docs/providers/openrouter.md @@ -51,8 +51,8 @@ This approach provides better flexibility for managing configurations across dif | Model Name | Function Call | |---------------------------|-----------------------------------------------------| -| openrouter/openai/gpt-3.5-turbo | `completion('openrouter/openai/gpt-3.5-turbo', messages)` | `os.environ['OR_SITE_URL']`,`os.environ['OR_APP_NAME']`,`os.environ['OPENROUTER_API_KEY']` | -| openrouter/openai/gpt-3.5-turbo-16k | `completion('openrouter/openai/gpt-3.5-turbo-16k', messages)` | `os.environ['OR_SITE_URL']`,`os.environ['OR_APP_NAME']`,`os.environ['OPENROUTER_API_KEY']` | +| openrouter/openai/gpt-4o | `completion('openrouter/openai/gpt-4o', messages)` | `os.environ['OR_SITE_URL']`,`os.environ['OR_APP_NAME']`,`os.environ['OPENROUTER_API_KEY']` | +| openrouter/openai/gpt-4o-16k | `completion('openrouter/openai/gpt-4o-16k', messages)` | `os.environ['OR_SITE_URL']`,`os.environ['OR_APP_NAME']`,`os.environ['OPENROUTER_API_KEY']` | | openrouter/openai/gpt-4 | `completion('openrouter/openai/gpt-4', messages)` | `os.environ['OR_SITE_URL']`,`os.environ['OR_APP_NAME']`,`os.environ['OPENROUTER_API_KEY']` | | openrouter/openai/gpt-4-32k | `completion('openrouter/openai/gpt-4-32k', messages)` | `os.environ['OR_SITE_URL']`,`os.environ['OR_APP_NAME']`,`os.environ['OPENROUTER_API_KEY']` | | openrouter/anthropic/claude-2 | `completion('openrouter/anthropic/claude-2', messages)` | `os.environ['OR_SITE_URL']`,`os.environ['OR_APP_NAME']`,`os.environ['OPENROUTER_API_KEY']` | diff --git a/docs/my-website/docs/providers/text_completion_openai.md b/docs/my-website/docs/providers/text_completion_openai.md index d790c01fe0b..ab150f3908f 100644 --- a/docs/my-website/docs/providers/text_completion_openai.md +++ b/docs/my-website/docs/providers/text_completion_openai.md @@ -21,7 +21,7 @@ os.environ["OPENAI_API_KEY"] = "your-api-key" # openai call response = completion( - model = "gpt-3.5-turbo-instruct", + model = "gpt-4o-instruct", messages=[{ "content": "Hello, how are you?","role": "user"}] ) ``` @@ -43,20 +43,20 @@ export OPENAI_API_KEY="" ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: - model: openai/gpt-3.5-turbo # The `openai/` prefix will call openai.chat.completions.create + model: openai/gpt-4o # The `openai/` prefix will call openai.chat.completions.create api_key: os.environ/OPENAI_API_KEY - - model_name: gpt-3.5-turbo-instruct + - model_name: gpt-4o-instruct litellm_params: - model: text-completion-openai/gpt-3.5-turbo-instruct # The `text-completion-openai/` prefix will call openai.completions.create + model: text-completion-openai/gpt-4o-instruct # The `text-completion-openai/` prefix will call openai.completions.create api_key: os.environ/OPENAI_API_KEY ``` Use this to add all openai models with one API Key. **WARNING: This will not do any load balancing** -This means requests to `gpt-4`, `gpt-3.5-turbo` , `gpt-4-turbo-preview` will all go through this route +This means requests to `gpt-4`, `gpt-4o` , `gpt-4-turbo-preview` will all go through this route ```yaml model_list: @@ -69,7 +69,7 @@ model_list: ```bash -$ litellm --model gpt-3.5-turbo-instruct +$ litellm --model gpt-4o-instruct # Server running on http://0.0.0.0:4000 ``` @@ -87,7 +87,7 @@ $ litellm --model gpt-3.5-turbo-instruct curl --location 'http://0.0.0.0:4000/chat/completions' \ --header 'Content-Type: application/json' \ --data ' { - "model": "gpt-3.5-turbo-instruct", + "model": "gpt-4o-instruct", "messages": [ { "role": "user", @@ -108,7 +108,7 @@ client = openai.OpenAI( ) # request sent to model set on litellm proxy, `litellm --model` -response = client.chat.completions.create(model="gpt-3.5-turbo-instruct", messages = [ +response = client.chat.completions.create(model="gpt-4o-instruct", messages = [ { "role": "user", "content": "this is a test request, write a short poem" @@ -132,7 +132,7 @@ from langchain.schema import HumanMessage, SystemMessage chat = ChatOpenAI( openai_api_base="http://0.0.0.0:4000", # set openai_api_base to the LiteLLM Proxy - model = "gpt-3.5-turbo-instruct", + model = "gpt-4o-instruct", temperature=0.1 ) @@ -156,8 +156,8 @@ print(response) | Model Name | Function Call | |---------------------|----------------------------------------------------| -| gpt-3.5-turbo-instruct | `response = completion(model="gpt-3.5-turbo-instruct", messages=messages)` | -| gpt-3.5-turbo-instruct-0914 | `response = completion(model="gpt-3.5-turbo-instruct-0914", messages=messages)` | +| gpt-4o-instruct | `response = completion(model="gpt-4o-instruct", messages=messages)` | +| gpt-4o-instruct-0914 | `response = completion(model="gpt-4o-instruct-0914", messages=messages)` | | text-davinci-003 | `response = completion(model="text-davinci-003", messages=messages)` | | ada-001 | `response = completion(model="ada-001", messages=messages)` | | curie-001 | `response = completion(model="curie-001", messages=messages)` |