diff --git a/docs/my-website/docs/index.md b/docs/my-website/docs/index.md index ca63c9e39ff..b67a265ba66 100644 --- a/docs/my-website/docs/index.md +++ b/docs/my-website/docs/index.md @@ -348,7 +348,7 @@ litellm --model huggingface/bigcode/starcoder ```yaml title="litellm_config.yaml" model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: model: azure/your-deployment api_base: os.environ/AZURE_API_BASE @@ -377,7 +377,7 @@ import openai client = openai.OpenAI(api_key="anything", base_url="http://0.0.0.0:4000") response = client.chat.completions.create( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{"role": "user", "content": "Write a short poem"}] ) print(response.choices[0].message.content) diff --git a/docs/my-website/docs/providers/openai.md b/docs/my-website/docs/providers/openai.md index 2907cdf9f47..c86c4d599cf 100644 --- a/docs/my-website/docs/providers/openai.md +++ b/docs/my-website/docs/providers/openai.md @@ -58,20 +58,20 @@ export OPENAI_API_KEY="" ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: - model: openai/gpt-3.5-turbo # The `openai/` prefix will call openai.chat.completions.create + model: openai/gpt-4o # The `openai/` prefix will call openai.chat.completions.create api_key: os.environ/OPENAI_API_KEY - - model_name: gpt-3.5-turbo-instruct + - model_name: gpt-4o-instruct litellm_params: - model: text-completion-openai/gpt-3.5-turbo-instruct # The `text-completion-openai/` prefix will call openai.completions.create + model: text-completion-openai/gpt-4o-instruct # The `text-completion-openai/` prefix will call openai.completions.create api_key: os.environ/OPENAI_API_KEY ``` Use this to add all openai models with one API Key. **WARNING: This will not do any load balancing** -This means requests to `gpt-4`, `gpt-3.5-turbo` , `gpt-4-turbo-preview` will all go through this route +This means requests to `gpt-4`, `gpt-4o` , `gpt-4-turbo-preview` will all go through this route ```yaml model_list: @@ -84,7 +84,7 @@ model_list: ```bash -$ litellm --model gpt-3.5-turbo +$ litellm --model gpt-4o # Server running on http://0.0.0.0:4000 ``` @@ -102,7 +102,7 @@ $ litellm --model gpt-3.5-turbo curl --location 'http://0.0.0.0:4000/chat/completions' \ --header 'Content-Type: application/json' \ --data ' { - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", @@ -123,7 +123,7 @@ client = openai.OpenAI( ) # request sent to model set on litellm proxy, `litellm --model` -response = client.chat.completions.create(model="gpt-3.5-turbo", messages = [ +response = client.chat.completions.create(model="gpt-4o", messages = [ { "role": "user", "content": "this is a test request, write a short poem" @@ -147,7 +147,7 @@ from langchain.schema import HumanMessage, SystemMessage chat = ChatOpenAI( openai_api_base="http://0.0.0.0:4000", # set openai_api_base to the LiteLLM Proxy - model = "gpt-3.5-turbo", + model = "gpt-4o", temperature=0.1 ) @@ -219,12 +219,12 @@ os.environ["OPENAI_BASE_URL"] = "https://your_host/v1" # OPTIONAL | gpt-4-turbo-preview | `response = completion(model="gpt-4-0125-preview", messages=messages)` | | gpt-4-0125-preview | `response = completion(model="gpt-4-0125-preview", messages=messages)` | | gpt-4-1106-preview | `response = completion(model="gpt-4-1106-preview", messages=messages)` | -| gpt-3.5-turbo-1106 | `response = completion(model="gpt-3.5-turbo-1106", messages=messages)` | -| gpt-3.5-turbo | `response = completion(model="gpt-3.5-turbo", messages=messages)` | -| gpt-3.5-turbo-0301 | `response = completion(model="gpt-3.5-turbo-0301", messages=messages)` | -| gpt-3.5-turbo-0613 | `response = completion(model="gpt-3.5-turbo-0613", messages=messages)` | -| gpt-3.5-turbo-16k | `response = completion(model="gpt-3.5-turbo-16k", messages=messages)` | -| gpt-3.5-turbo-16k-0613| `response = completion(model="gpt-3.5-turbo-16k-0613", messages=messages)` | +| gpt-4o-1106 | `response = completion(model="gpt-4o-1106", messages=messages)` | +| gpt-4o | `response = completion(model="gpt-4o", messages=messages)` | +| gpt-4o-0301 | `response = completion(model="gpt-4o-0301", messages=messages)` | +| gpt-4o-0613 | `response = completion(model="gpt-4o-0613", messages=messages)` | +| gpt-4o-16k | `response = completion(model="gpt-4o-16k", messages=messages)` | +| gpt-4o-16k-0613| `response = completion(model="gpt-4o-16k-0613", messages=messages)` | | gpt-4 | `response = completion(model="gpt-4", messages=messages)` | | gpt-4-0314 | `response = completion(model="gpt-4-0314", messages=messages)` | | gpt-4-0613 | `response = completion(model="gpt-4-0613", messages=messages)` | @@ -428,9 +428,9 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \ |---------------------------|-----------------------------------------------------------------| | fine tuned `gpt-4-0613` | `response = completion(model="ft:gpt-4-0613", messages=messages)` | | fine tuned `gpt-4o-2024-05-13` | `response = completion(model="ft:gpt-4o-2024-05-13", messages=messages)` | -| fine tuned `gpt-3.5-turbo-0125` | `response = completion(model="ft:gpt-3.5-turbo-0125", messages=messages)` | -| fine tuned `gpt-3.5-turbo-1106` | `response = completion(model="ft:gpt-3.5-turbo-1106", messages=messages)` | -| fine tuned `gpt-3.5-turbo-0613` | `response = completion(model="ft:gpt-3.5-turbo-0613", messages=messages)` | +| fine tuned `gpt-4o-0125` | `response = completion(model="ft:gpt-4o-0125", messages=messages)` | +| fine tuned `gpt-4o-1106` | `response = completion(model="ft:gpt-4o-1106", messages=messages)` | +| fine tuned `gpt-4o-0613` | `response = completion(model="ft:gpt-4o-0613", messages=messages)` | ## Getting Reasoning Content in `/chat/completions` @@ -993,7 +993,7 @@ tools = [ ] response = litellm.completion( - model="gpt-3.5-turbo-1106", + model="gpt-4o-1106", messages=messages, tools=tools, tool_choice="auto", # auto is default, but we'll be explicit @@ -1011,7 +1011,7 @@ from litellm import completion os.environ["OPENAI_API_KEY"] = "your-api-key" response = completion( - model = "gpt-3.5-turbo", + model = "gpt-4o", messages=[{ "content": "Hello, how are you?","role": "user"}], extra_headers={"AI-Resource Group": "ishaan-resource"} ) @@ -1030,7 +1030,7 @@ os.environ["OPENAI_API_KEY"] = "your-api-key" os.environ["OPENAI_ORGANIZATION"] = "your-org-id" # OPTIONAL response = completion( - model = "gpt-3.5-turbo", + model = "gpt-4o", messages=[{ "content": "Hello, how are you?","role": "user"}] ) ``` @@ -1047,14 +1047,14 @@ import litellm, httpx # for completion litellm.client_session = httpx.Client(verify=False) response = litellm.completion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=messages, ) # for acompletion litellm.aclient_session = httpx.AsyncClient(verify=False) response = litellm.acompletion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=messages, ) ``` @@ -1092,9 +1092,9 @@ Forward openai Org ID's from the client to OpenAI with `forward_openai_org_id` p ```yaml model_list: - - model_name: "gpt-3.5-turbo" + - model_name: "gpt-4o" litellm_params: - model: gpt-3.5-turbo + model: gpt-4o api_key: os.environ/OPENAI_API_KEY general_settings: @@ -1119,7 +1119,7 @@ client = OpenAI( base_url="http://0.0.0.0:4000" ) -client.chat.completions.create(model="gpt-3.5-turbo", messages=[{"role": "user", "content": "Hello world"}]) +client.chat.completions.create(model="gpt-4o", messages=[{"role": "user", "content": "Hello world"}]) ``` In your logs you should see the forwarded org id diff --git a/docs/my-website/docs/proxy/quick_start.md b/docs/my-website/docs/proxy/quick_start.md index cf1ab78b352..b37ebeb1c44 100644 --- a/docs/my-website/docs/proxy/quick_start.md +++ b/docs/my-website/docs/proxy/quick_start.md @@ -40,7 +40,7 @@ In a new shell, run, this will make an `openai.chat.completions` request. Ensure litellm --test ``` -This will now automatically route any requests for gpt-3.5-turbo to bigcode starcoder, hosted on huggingface inference endpoints. +This will now automatically route any requests for gpt-4o to bigcode starcoder, hosted on huggingface inference endpoints. ### Supported LLMs All LiteLLM supported LLMs are supported on the Proxy. Seel all [supported llms](https://docs.litellm.ai/docs/providers) @@ -75,7 +75,7 @@ $ export OPENAI_API_KEY=my-api-key ``` ```shell -$ litellm --model gpt-3.5-turbo +$ litellm --model gpt-4o ``` @@ -231,12 +231,12 @@ Example config ```yaml model_list: - - model_name: gpt-3.5-turbo # user-facing model alias + - model_name: gpt-4o # user-facing model alias litellm_params: # all params accepted by litellm.completion() - https://docs.litellm.ai/docs/completion/input model: azure/ api_base: api_key: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: model: azure/gpt-turbo-small-ca api_base: https://my-endpoint-canada-berri992.openai.azure.com/ @@ -270,7 +270,7 @@ LiteLLM is compatible with several SDKs - including OpenAI SDK, Anthropic SDK, M curl --location 'http://0.0.0.0:4000/chat/completions' \ --header 'Content-Type: application/json' \ --data ' { - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", @@ -291,7 +291,7 @@ client = openai.OpenAI( ) # request sent to model set on litellm proxy, `litellm --model` -response = client.chat.completions.create(model="gpt-3.5-turbo", messages = [ +response = client.chat.completions.create(model="gpt-4o", messages = [ { "role": "user", "content": "this is a test request, write a short poem" @@ -315,7 +315,7 @@ from langchain.schema import HumanMessage, SystemMessage chat = ChatOpenAI( openai_api_base="http://0.0.0.0:4000", # set openai_api_base to the LiteLLM Proxy - model = "gpt-3.5-turbo", + model = "gpt-4o", temperature=0.1 ) @@ -375,7 +375,7 @@ This is **not recommended**. There is duplicate logic as the proxy also uses the from litellm import completion response = completion( - model="openai/gpt-3.5-turbo", + model="openai/gpt-4o", messages = [ { "role": "user", @@ -436,12 +436,12 @@ print(message.content) Events that occur during normal operation ```shell -litellm --model gpt-3.5-turbo --debug +litellm --model gpt-4o --debug ``` Detailed information ```shell -litellm --model gpt-3.5-turbo --detailed_debug +litellm --model gpt-4o --detailed_debug ``` ### Set Debug Level using env variables diff --git a/docs/my-website/docs/proxy_server.md b/docs/my-website/docs/proxy_server.md index e23d64e443b..3c410a49d00 100644 --- a/docs/my-website/docs/proxy_server.md +++ b/docs/my-website/docs/proxy_server.md @@ -148,7 +148,7 @@ openai.api_key = "any-string-here" openai.api_base = "http://0.0.0.0:8080" # your proxy url # call openai -response = openai.ChatCompletion.create(model="gpt-3.5-turbo", messages=[{"role": "user", "content": "Hey"}]) +response = openai.ChatCompletion.create(model="gpt-4o", messages=[{"role": "user", "content": "Hey"}]) print(response) @@ -409,8 +409,8 @@ import openai openai.api_key = "any-string-here" openai.api_base = "http://0.0.0.0:8080" # your proxy url -# call gpt-3.5-turbo -response = openai.ChatCompletion.create(model="gpt-3.5-turbo", messages=[{"role": "user", "content": "Hey"}]) +# call gpt-4o +response = openai.ChatCompletion.create(model="gpt-4o", messages=[{"role": "user", "content": "Hey"}]) print(response) diff --git a/docs/my-website/docs/set_keys.md b/docs/my-website/docs/set_keys.md index 295d9ec5501..2354afa862a 100644 --- a/docs/my-website/docs/set_keys.md +++ b/docs/my-website/docs/set_keys.md @@ -74,7 +74,7 @@ This variable is checked for all providers import litellm # openai call litellm.api_key = "sk-OpenAIKey" -response = litellm.completion(messages=messages, model="gpt-3.5-turbo") +response = litellm.completion(messages=messages, model="gpt-4o") # anthropic call litellm.api_key = "sk-AnthropicKey" @@ -85,7 +85,7 @@ response = litellm.completion(messages=messages, model="claude-2") ```python litellm.openai_key = "sk-OpenAIKey" -response = litellm.completion(messages=messages, model="gpt-3.5-turbo") +response = litellm.completion(messages=messages, model="gpt-4o") # anthropic call litellm.anthropic_key = "sk-AnthropicKey" @@ -97,7 +97,7 @@ response = litellm.completion(messages=messages, model="claude-2") ```python import litellm litellm.api_base = "https://hosted-llm-api.co" -response = litellm.completion(messages=messages, model="gpt-3.5-turbo") +response = litellm.completion(messages=messages, model="gpt-4o") ``` ### litellm.api_version @@ -105,14 +105,14 @@ response = litellm.completion(messages=messages, model="gpt-3.5-turbo") ```python import litellm litellm.api_version = "2023-05-15" -response = litellm.completion(messages=messages, model="gpt-3.5-turbo") +response = litellm.completion(messages=messages, model="gpt-4o") ``` ### litellm.organization ```python import litellm litellm.organization = "LiteLlmOrg" -response = litellm.completion(messages=messages, model="gpt-3.5-turbo") +response = litellm.completion(messages=messages, model="gpt-4o") ``` ## Passing Args to completion() (or any litellm endpoint - `transcription`, `embedding`, `text_completion`, etc) @@ -156,7 +156,7 @@ Check if a user submitted a valid key for the model they're trying to call. ```python key = "bad-key" -response = check_valid_key(model="gpt-3.5-turbo", api_key=key) +response = check_valid_key(model="gpt-4o", api_key=key) assert(response == False) ``` @@ -217,5 +217,5 @@ This helper tells you if you have all the required environment variables for a m ```python from litellm import validate_environment -print(validate_environment("openai/gpt-3.5-turbo")) +print(validate_environment("openai/gpt-4o")) ``` \ No newline at end of file diff --git a/docs/my-website/static/llms-full.txt b/docs/my-website/static/llms-full.txt index 203dfd12bab..bea59af8727 100644 --- a/docs/my-website/static/llms-full.txt +++ b/docs/my-website/static/llms-full.txt @@ -73,7 +73,7 @@ import os os.environ["OPENAI_API_KEY"] = "your-api-key" response = completion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{ "content": "Hello, how are you?","role": "user"}] ) @@ -218,7 +218,7 @@ import os os.environ["OPENAI_API_KEY"] = "your-api-key" response = completion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{ "content": "Hello, how are you?","role": "user"}], stream=True, ) @@ -386,7 +386,7 @@ os.environ["OPENAI_API_KEY"] litellm.success_callback = ["lunary", "mlflow", "langfuse", "helicone"] # log input/output to lunary, mlflow, langfuse, helicone #openai call -response = completion(model="gpt-3.5-turbo", messages=[{"role": "user", "content": "Hi 👋 - i'm openai"}]) +response = completion(model="gpt-4o", messages=[{"role": "user", "content": "Hi 👋 - i'm openai"}]) ``` @@ -413,7 +413,7 @@ litellm.success_callback = [track_cost_callback] # set custom callback function # litellm.completion() call response = completion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[\ {\ "role": "user",\ @@ -467,7 +467,7 @@ Example `litellm_config.yaml` ```codeBlockLines_e6Vv model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: model: azure/ api_base: os.environ/AZURE_API_BASE # runs os.getenv("AZURE_API_BASE") @@ -495,7 +495,7 @@ docker run \ import openai # openai v1.0.0+ client = openai.OpenAI(api_key="anything",base_url="http://0.0.0.0:4000") # set proxy to base_url # request sent to model set on litellm proxy, `litellm --model` -response = client.chat.completions.create(model="gpt-3.5-turbo", messages = [\ +response = client.chat.completions.create(model="gpt-4o", messages = [\ {\ "role": "user",\ "content": "this is a test request, write a short poem"\ @@ -621,9 +621,9 @@ Here's the exact json output you can expect from a litellm `completion` call: | Model Name | Function Call | Required OS Variables | | --- | --- | --- | -| gpt-3.5-turbo | `completion('gpt-3.5-turbo', messages)` | `os.environ['OPENAI_API_KEY']` | -| gpt-3.5-turbo-16k | `completion('gpt-3.5-turbo-16k', messages)` | `os.environ['OPENAI_API_KEY']` | -| gpt-3.5-turbo-16k-0613 | `completion('gpt-3.5-turbo-16k-0613', messages)` | `os.environ['OPENAI_API_KEY']` | +| gpt-4o | `completion('gpt-4o', messages)` | `os.environ['OPENAI_API_KEY']` | +| gpt-4o-16k | `completion('gpt-4o-16k', messages)` | `os.environ['OPENAI_API_KEY']` | +| gpt-4o-16k-0613 | `completion('gpt-4o-16k-0613', messages)` | `os.environ['OPENAI_API_KEY']` | | gpt-4 | `completion('gpt-4', messages)` | `os.environ['OPENAI_API_KEY']` | ## Azure OpenAI Chat Completion Models [​](https://docs.litellm.ai/completion/supported\#azure-openai-chat-completion-models "Direct link to Azure OpenAI Chat Completion Models") @@ -632,7 +632,7 @@ For Azure calls add the `azure/` prefix to `model`. If your azure deployment nam | Model Name | Function Call | Required OS Variables | | --- | --- | --- | -| gpt-3.5-turbo | `completion('azure/gpt-3.5-turbo-deployment', messages)` | `os.environ['AZURE_API_KEY']`, `os.environ['AZURE_API_BASE']`, `os.environ['AZURE_API_VERSION']` | +| gpt-4o | `completion('azure/gpt-4o-deployment', messages)` | `os.environ['AZURE_API_KEY']`, `os.environ['AZURE_API_BASE']`, `os.environ['AZURE_API_VERSION']` | | gpt-4 | `completion('azure/gpt-4-deployment', messages)` | `os.environ['AZURE_API_KEY']`, `os.environ['AZURE_API_BASE']`, `os.environ['AZURE_API_VERSION']` | ### OpenAI Text Completion Models [​](https://docs.litellm.ai/completion/supported\#openai-text-completion-models "Direct link to OpenAI Text Completion Models") @@ -678,8 +678,8 @@ All the text models from [OpenRouter](https://openrouter.ai/docs) are supported | Model Name | Function Call | Required OS Variables | | --- | --- | --- | -| openai/gpt-3.5-turbo | `completion('openai/gpt-3.5-turbo', messages)` | `os.environ['OR_SITE_URL']`, `os.environ['OR_APP_NAME']`, `os.environ['OR_API_KEY']` | -| openai/gpt-3.5-turbo-16k | `completion('openai/gpt-3.5-turbo-16k', messages)` | `os.environ['OR_SITE_URL']`, `os.environ['OR_APP_NAME']`, `os.environ['OR_API_KEY']` | +| openai/gpt-4o | `completion('openai/gpt-4o', messages)` | `os.environ['OR_SITE_URL']`, `os.environ['OR_APP_NAME']`, `os.environ['OR_API_KEY']` | +| openai/gpt-4o-16k | `completion('openai/gpt-4o-16k', messages)` | `os.environ['OR_SITE_URL']`, `os.environ['OR_APP_NAME']`, `os.environ['OR_API_KEY']` | | openai/gpt-4 | `completion('openai/gpt-4', messages)` | `os.environ['OR_SITE_URL']`, `os.environ['OR_APP_NAME']`, `os.environ['OR_API_KEY']` | | openai/gpt-4-32k | `completion('openai/gpt-4-32k', messages)` | `os.environ['OR_SITE_URL']`, `os.environ['OR_APP_NAME']`, `os.environ['OR_API_KEY']` | | anthropic/claude-2 | `completion('anthropic/claude-2', messages)` | `os.environ['OR_SITE_URL']`, `os.environ['OR_APP_NAME']`, `os.environ['OR_API_KEY']` | @@ -874,7 +874,7 @@ os.environ['SENTRY_DSN'], os.environ['SENTRY_API_TRACE_RATE']= "" os.environ['POSTHOG_API_KEY'], os.environ['POSTHOG_API_URL'] = "api-key", "api-url" os.environ["HELICONE_API_KEY"] = "" -response = completion(model="gpt-3.5-turbo", messages=messages) +response = completion(model="gpt-4o", messages=messages) ``` @@ -916,7 +916,7 @@ os.environ["OPENAI_API_KEY"], os.environ["COHERE_API_KEY"] = "", "" litellm.success_callback=["helicone"] #openai call -response = completion(model="gpt-3.5-turbo", messages=[{"role": "user", "content": "Hi 👋 - i'm openai"}]) +response = completion(model="gpt-4o", messages=[{"role": "user", "content": "Hi 👋 - i'm openai"}]) #cohere call response = completion(model="command-nightly", messages=[{"role": "user", "content": "Hi 👋 - i'm cohere"}]) @@ -942,7 +942,7 @@ litellm.api_base = "https://oai.hconeai.com/v1" litellm.headers = {"Helicone-Auth": f"Bearer {os.getenv('HELICONE_API_KEY')}"} response = litellm.completion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{"role": "user", "content": "how does a court case get to the Supreme Court?"}] ) @@ -1020,7 +1020,7 @@ litellm.success_callback=["supabase"] litellm.failure_callback=["supabase"] #openai call -response = completion(model="gpt-3.5-turbo", messages=[{"role": "user", "content": "Hi 👋 - i'm openai"}]) +response = completion(model="gpt-4o", messages=[{"role": "user", "content": "Hi 👋 - i'm openai"}]) #bad call response = completion(model="chatgpt-test", messages=[{"role": "user", "content": "Hi 👋 - i'm a bad call to test error logging"}]) @@ -3340,7 +3340,7 @@ curl -i http://localhost:4000/v1/chat/completions \ -H "Content-Type: application/json" \ -H "Authorization: Bearer sk-npnwjPQciVRok5yNZgKmFQ" \ -d '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [\ {"role": "user", "content": "hi my email is ishaan@berri.ai"}\ ], @@ -5727,7 +5727,7 @@ curl -i http://localhost:4000/v1/chat/completions \ -H "Content-Type: application/json" \ -H "Authorization: Bearer sk-npnwjPQciVRok5yNZgKmFQ" \ -d '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [\ {"role": "user", "content": "hi my email is ishaan@berri.ai"}\ ], @@ -6605,7 +6605,7 @@ curl -i http://localhost:4000/v1/chat/completions \ -H "Content-Type: application/json" \ -H "Authorization: Bearer sk-npnwjPQciVRok5yNZgKmFQ" \ -d '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [\ {"role": "user", "content": "hi my email is ishaan@berri.ai"}\ ], @@ -7071,7 +7071,7 @@ curl -i http://localhost:4000/v1/chat/completions \ -H "Content-Type: application/json" \ -H "Authorization: Bearer sk-npnwjPQciVRok5yNZgKmFQ" \ -d '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [\ {"role": "user", "content": "hi my email is ishaan@berri.ai"}\ ],