diff --git a/docs/my-website/docs/tutorials/azure_openai.md b/docs/my-website/docs/tutorials/azure_openai.md index 16436550a0d..cd984caaf88 100644 --- a/docs/my-website/docs/tutorials/azure_openai.md +++ b/docs/my-website/docs/tutorials/azure_openai.md @@ -24,7 +24,7 @@ os.environ["AZURE_API_VERSION"] = "2023-05-15" # openai call response = completion( - model = "gpt-3.5-turbo", + model = "gpt-4o", messages = [{ "content": "Hello, how are you?","role": "user"}] ) print("Openai Response\n") @@ -56,7 +56,7 @@ os.environ["AZURE_API_VERSION"] = "2023-05-15" # openai call response = completion( - model = "gpt-3.5-turbo", + model = "gpt-4o", messages = [{ "content": "Hello, how are you?","role": "user"}], stream=True ) @@ -93,7 +93,7 @@ os.environ["AZURE_API_VERSION"] = "2023-05-15" # openai call response = acompletion( - model = "gpt-3.5-turbo", + model = "gpt-4o", messages = [{ "content": "Hello, how are you?","role": "user"}], stream=True ) @@ -132,7 +132,7 @@ os.environ["AZURE_API_KEY"] = "YOUR_AZURE_API_KEY" messages = [{"content": "Hello, how are you?", "role": "user"}] # Create threads for making the completions -thread1 = threading.Thread(target=make_completion, args=("gpt-3.5-turbo", messages)) +thread1 = threading.Thread(target=make_completion, args=("gpt-4o", messages)) thread2 = threading.Thread(target=make_completion, args=("azure/your-azure-deployment", messages)) # Start both threads diff --git a/docs/my-website/docs/tutorials/compare_llms.md b/docs/my-website/docs/tutorials/compare_llms.md index 02877b46607..9bf6cb9a310 100644 --- a/docs/my-website/docs/tutorials/compare_llms.md +++ b/docs/my-website/docs/tutorials/compare_llms.md @@ -33,7 +33,7 @@ Supported LLMs: https://docs.litellm.ai/docs/providers ```python # Define the list of models to benchmark -models = ['gpt-3.5-turbo', 'claude-2'] +models = ['gpt-4o', 'claude-2'] # Enter LLM API keys os.environ['OPENAI_API_KEY'] = "" @@ -60,7 +60,7 @@ Benchmark Results for 'When will BerriAI IPO?': +-----------------+----------------------------------------------------------------------------------+---------------------------+------------+ | Model | Response | Response Time (seconds) | Cost ($) | +=================+==================================================================================+===========================+============+ -| gpt-3.5-turbo | As an AI language model, I cannot provide up-to-date information or predict | 1.55 seconds | $0.000122 | +| gpt-4o | As an AI language model, I cannot provide up-to-date information or predict | 1.55 seconds | $0.000122 | | | future events. It is best to consult a reliable financial source or contact | | | | | BerriAI directly for information regarding their IPO plans. | | | +-----------------+----------------------------------------------------------------------------------+---------------------------+------------+ @@ -109,7 +109,7 @@ os.environ["OPENAI_API_KEY"] = "openai key" os.environ["COHERE_API_KEY"] = "cohere key" messages = [{ "content": "Hello, how are you?","role": "user"}] # openai call -response = completion(model="gpt-3.5-turbo", messages=messages) +response = completion(model="gpt-4o", messages=messages) # cohere call response = completion("command-nightly", messages) ``` @@ -151,7 +151,7 @@ os.environ['ANTHROPIC_API_KEY'] = "" # using https://api.together.xyz/playground for llama2 # try any supported LLM here: https://docs.litellm.ai/docs/providers -models = ['togethercomputer/llama-2-70b-chat', 'gpt-3.5-turbo', 'claude-instant-1.2'] +models = ['togethercomputer/llama-2-70b-chat', 'gpt-4o', 'claude-instant-1.2'] data = [] for question in questions: # group by question @@ -227,7 +227,7 @@ for question, group_data in grouped_by_question: 1 - gpt-3.5-turbo + gpt-4o what is litellm? litellm is a light package that provides a simplified interface for making API calls to various language models and APIs. It abstracts away the complexities of handling network requests, authentication, and response parsing, making it easier for developers to integrate powerful language models into their applications.\n\nWith litellm, you can quickly make API calls to models like OpenAI's GPT-3.5 Turbo, Azure's Text Analytics, Cohere's Command API, Anthropic's API, and Huggingface's models. It also supports additional functionality like conversational AI, summarization, translation, and more.\n\nBy using litellm, you can focus on your application logic without getting tangled in the details of API integration, allowing you to quickly build intelligent and conversational applications. 8.324332 @@ -276,7 +276,7 @@ os.environ['TOGETHERAI_API_KEY'] = "" os.environ['OPENAI_API_KEY'] = "" os.environ['ANTHROPIC_API_KEY'] = "" -models = ['togethercomputer/llama-2-70b-chat', 'gpt-3.5-turbo', 'claude-instant-1.2'] # enter llms to benchmark +models = ['togethercomputer/llama-2-70b-chat', 'gpt-4o', 'claude-instant-1.2'] # enter llms to benchmark data_2 = [] for question in questions: # group by question @@ -354,7 +354,7 @@ for question, group_data in grouped_by_question: 4 - gpt-3.5-turbo + gpt-4o User input: Hi, I'm [your name] and I'm excited about using LiteLLM to simplify working with different LLM providers. Before finding LiteLLM, I faced challenges working with multiple LLMs. With LiteLLM's unified API and automatic translation, I believe it will help me achieve my goals of [state your goals]. I look forward to being part of this community and learning how to build impactful applications with LLMs. Let me know if you need any further clarification or details. 7.385472 0.000525 diff --git a/docs/my-website/docs/tutorials/compare_llms_2.md b/docs/my-website/docs/tutorials/compare_llms_2.md index 20aee68890d..ab37359567f 100644 --- a/docs/my-website/docs/tutorials/compare_llms_2.md +++ b/docs/my-website/docs/tutorials/compare_llms_2.md @@ -6,7 +6,7 @@ import Image from '@theme/IdealImage';
LiteLLM allows you to use any LLM as a drop in replacement for -`gpt-3.5-turbo` +`gpt-4o` This notebook walks through how you can compare GPT-4 vs Claude-2 on a given test set using litellm @@ -65,7 +65,7 @@ os.environ['ANTHROPIC_API_KEY'] = ""
-## Calling gpt-3.5-turbo and claude-2 on the same questions +## Calling gpt-4o and claude-2 on the same questions ## LiteLLM `completion()` allows you to call all LLMs in the same format @@ -76,7 +76,7 @@ os.environ['ANTHROPIC_API_KEY'] = "" ``` python results = [] # for storing results -models = ['gpt-3.5-turbo', 'claude-2'] # define what models you're testing, see: https://docs.litellm.ai/docs/providers +models = ['gpt-4o', 'claude-2'] # define what models you're testing, see: https://docs.litellm.ai/docs/providers for question in questions: row = [question] for model in models: diff --git a/docs/my-website/docs/tutorials/eval_suites.md b/docs/my-website/docs/tutorials/eval_suites.md index b533da99367..d3c6638c4da 100644 --- a/docs/my-website/docs/tutorials/eval_suites.md +++ b/docs/my-website/docs/tutorials/eval_suites.md @@ -235,7 +235,7 @@ pip install autoevals ### Quick Start In this code sample we use the `Factuality()` evaluator from `autoevals.llm` to test whether an output is factual, compared to an original (expected) value. -**Autoevals uses gpt-3.5-turbo / gpt-4-turbo by default to evaluate responses** +**Autoevals uses gpt-4o / gpt-4-turbo by default to evaluate responses** See autoevals docs on the [supported evaluators](https://www.braintrustdata.com/docs/autoevals/python#autoevalsllm) - Translation, Summary, Security Evaluators etc @@ -248,7 +248,7 @@ import litellm # litellm completion call question = "which country has the highest population" response = litellm.completion( - model = "gpt-3.5-turbo", + model = "gpt-4o", messages = [ { "role": "user", diff --git a/docs/my-website/docs/tutorials/fallbacks.md b/docs/my-website/docs/tutorials/fallbacks.md index 3c6c5b6bc73..c607064fb49 100644 --- a/docs/my-website/docs/tutorials/fallbacks.md +++ b/docs/my-website/docs/tutorials/fallbacks.md @@ -12,7 +12,7 @@ To use fallback models with `completion()`, specify a list of models in the `fal The `fallbacks` list should include the primary model you want to use, followed by additional models that can be used as backups in case the primary model fails to provide a response. ```python -response = completion(model="bad-model", fallbacks=["gpt-3.5-turbo" "command-nightly"], messages=messages) +response = completion(model="bad-model", fallbacks=["gpt-4o" "command-nightly"], messages=messages) ``` ## How does `completion_with_fallbacks()` work @@ -25,12 +25,12 @@ Completion with 'bad-model': got exception Unable to map your input to a model. -completion call gpt-3.5-turbo +completion call gpt-4o { "id": "chatcmpl-7qTmVRuO3m3gIBg4aTmAumV1TmQhB", "object": "chat.completion", "created": 1692741891, - "model": "gpt-3.5-turbo-0613", + "model": "gpt-4o-0613", "choices": [ { "index": 0, diff --git a/docs/my-website/docs/tutorials/finetuned_chat_gpt.md b/docs/my-website/docs/tutorials/finetuned_chat_gpt.md index 5dde3b3ff94..8ff4516f9d4 100644 --- a/docs/my-website/docs/tutorials/finetuned_chat_gpt.md +++ b/docs/my-website/docs/tutorials/finetuned_chat_gpt.md @@ -1,6 +1,6 @@ -# Using Fine-Tuned gpt-3.5-turbo -LiteLLM allows you to call `completion` with your fine-tuned gpt-3.5-turbo models -If you're trying to create your custom fine-tuned gpt-3.5-turbo model following along on this tutorial: https://platform.openai.com/docs/guides/fine-tuning/preparing-your-dataset +# Using Fine-Tuned gpt-4o +LiteLLM allows you to call `completion` with your fine-tuned gpt-4o models +If you're trying to create your custom fine-tuned gpt-4o model following along on this tutorial: https://platform.openai.com/docs/guides/fine-tuning/preparing-your-dataset Once you've created your fine-tuned model, you can call it with `litellm.completion()` @@ -13,7 +13,7 @@ from litellm import completion os.environ["OPENAI_API_KEY"] = "your-api-key" response = completion( - model="ft:gpt-3.5-turbo:my-org:custom_suffix:id", + model="ft:gpt-4o:my-org:custom_suffix:id", messages=[ {"role": "system", "content": "You are a helpful assistant."}, {"role": "user", "content": "Hello!"} @@ -39,7 +39,7 @@ os.environ["OPENAI_API_KEY"] = "your-api-key" os.environ["OPENAI_ORGANIZATION"] = "your-org-id" # Optional response = completion( - model="ft:gpt-3.5-turbo:my-org:custom_suffix:id", + model="ft:gpt-4o:my-org:custom_suffix:id", messages=[ {"role": "system", "content": "You are a helpful assistant."}, {"role": "user", "content": "Hello!"} diff --git a/docs/my-website/docs/tutorials/first_playground.md b/docs/my-website/docs/tutorials/first_playground.md index bc34e89b6c2..59cebdf9cce 100644 --- a/docs/my-website/docs/tutorials/first_playground.md +++ b/docs/my-website/docs/tutorials/first_playground.md @@ -39,7 +39,7 @@ os.environ["AI21_API_KEY"] = "ai21 key" ## REPLACE THIS messages = [{ "content": "Hello, how are you?","role": "user"}] # openai call -response = completion(model="gpt-3.5-turbo", messages=messages) +response = completion(model="gpt-4o", messages=messages) # cohere call response = completion("command-nightly", messages) @@ -130,7 +130,7 @@ Run this curl command to test it: curl -X POST localhost:4000/chat/completions \ -H 'Content-Type: application/json' \ -d '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [{ "content": "Hello, how are you?", "role": "user" diff --git a/docs/my-website/docs/tutorials/litellm_Test_Multiple_Providers.md b/docs/my-website/docs/tutorials/litellm_Test_Multiple_Providers.md index 2503e3cbf6f..ca2ca66ec2f 100644 --- a/docs/my-website/docs/tutorials/litellm_Test_Multiple_Providers.md +++ b/docs/my-website/docs/tutorials/litellm_Test_Multiple_Providers.md @@ -34,7 +34,7 @@ In this example, let's ask some questions about Paul Graham ```python -models = ["gpt-3.5-turbo", "gpt-3.5-turbo-16k", "gpt-4", "claude-instant-1", "replicate/llama-2-70b-chat:58d078176e02c219e11eb4da5a02a7830a283b14cf8f94537af893ccff5ee781"] +models = ["gpt-4o", "gpt-4o-16k", "gpt-4", "claude-instant-1", "replicate/llama-2-70b-chat:58d078176e02c219e11eb4da5a02a7830a283b14cf8f94537af893ccff5ee781"] context = """Paul Graham (/ɡræm/; born 1964)[3] is an English computer scientist, essayist, entrepreneur, venture capitalist, and author. He is best known for his work on the programming language Lisp, his former startup Viaweb (later renamed Yahoo! Store), cofounding the influential startup accelerator and seed capital firm Y Combinator, his essays, and Hacker News. He is the author of several computer programming books, including: On Lisp,[4] ANSI Common Lisp,[5] and Hackers & Painters.[6] Technology journalist Steven Levy has described Graham as a "hacker philosopher".[7] Graham was born in England, where he and his family maintain permanent residence. However he is also a citizen of the United States, where he was educated, lived, and worked until 2016.""" prompts = ["Who is Paul Graham?", "What is Paul Graham known for?" , "Is paul graham a writer?" , "Where does Paul Graham live?", "What has Paul Graham done?"] messages = [[{"role": "user", "content": context + "\n" + prompt}] for prompt in prompts] # pass in a list of messages we want to test @@ -48,7 +48,7 @@ Run 100+ simultaneous queries across multiple providers to see when they fail + ```python -models=["gpt-3.5-turbo", "replicate/llama-2-70b-chat:58d078176e02c219e11eb4da5a02a7830a283b14cf8f94537af893ccff5ee781", "claude-instant-1"] +models=["gpt-4o", "replicate/llama-2-70b-chat:58d078176e02c219e11eb4da5a02a7830a283b14cf8f94537af893ccff5ee781", "claude-instant-1"] context = """Paul Graham (/ɡræm/; born 1964)[3] is an English computer scientist, essayist, entrepreneur, venture capitalist, and author. He is best known for his work on the programming language Lisp, his former startup Viaweb (later renamed Yahoo! Store), cofounding the influential startup accelerator and seed capital firm Y Combinator, his essays, and Hacker News. He is the author of several computer programming books, including: On Lisp,[4] ANSI Common Lisp,[5] and Hackers & Painters.[6] Technology journalist Steven Levy has described Graham as a "hacker philosopher".[7] Graham was born in England, where he and his family maintain permanent residence. However he is also a citizen of the United States, where he was educated, lived, and worked until 2016.""" prompt = "Where does Paul Graham live?" final_prompt = context + prompt @@ -95,7 +95,7 @@ Run load testing for 2 mins. Hitting endpoints with 100+ queries every 15 second ```python -models=["gpt-3.5-turbo", "replicate/llama-2-70b-chat:58d078176e02c219e11eb4da5a02a7830a283b14cf8f94537af893ccff5ee781", "claude-instant-1"] +models=["gpt-4o", "replicate/llama-2-70b-chat:58d078176e02c219e11eb4da5a02a7830a283b14cf8f94537af893ccff5ee781", "claude-instant-1"] context = """Paul Graham (/ɡræm/; born 1964)[3] is an English computer scientist, essayist, entrepreneur, venture capitalist, and author. He is best known for his work on the programming language Lisp, his former startup Viaweb (later renamed Yahoo! Store), cofounding the influential startup accelerator and seed capital firm Y Combinator, his essays, and Hacker News. He is the author of several computer programming books, including: On Lisp,[4] ANSI Common Lisp,[5] and Hackers & Painters.[6] Technology journalist Steven Levy has described Graham as a "hacker philosopher".[7] Graham was born in England, where he and his family maintain permanent residence. However he is also a citizen of the United States, where he was educated, lived, and worked until 2016.""" prompt = "Where does Paul Graham live?" final_prompt = context + prompt diff --git a/docs/my-website/docs/tutorials/litellm_proxy_aporia.md b/docs/my-website/docs/tutorials/litellm_proxy_aporia.md index 07eb36baa8b..1cfc2e4309c 100644 --- a/docs/my-website/docs/tutorials/litellm_proxy_aporia.md +++ b/docs/my-website/docs/tutorials/litellm_proxy_aporia.md @@ -37,9 +37,9 @@ Add the `Toxicity - Response` to your Post LLM API Call project - Define your guardrails under the `guardrails` section and set `pre_call_guardrails` and `post_call_guardrails` ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: - model: openai/gpt-3.5-turbo + model: openai/gpt-4o api_key: os.environ/OPENAI_API_KEY guardrails: @@ -84,7 +84,7 @@ curl -i http://localhost:4000/v1/chat/completions \ -H "Content-Type: application/json" \ -H "Authorization: Bearer sk-npnwjPQciVRok5yNZgKmFQ" \ -d '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ {"role": "user", "content": "hi my email is ishaan@berri.ai"} ], @@ -123,7 +123,7 @@ curl -i http://localhost:4000/v1/chat/completions \ -H "Content-Type: application/json" \ -H "Authorization: Bearer sk-npnwjPQciVRok5yNZgKmFQ" \ -d '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ {"role": "user", "content": "hi what is the weather"} ], @@ -180,7 +180,7 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \ --header 'Authorization: Bearer sk-jNm1Zar7XfNdZXp49Z1kSQ' \ --header 'Content-Type: application/json' \ --data '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", diff --git a/docs/my-website/docs/tutorials/lm_evaluation_harness.md b/docs/my-website/docs/tutorials/lm_evaluation_harness.md index 01fdb4b304c..a1cbd478612 100644 --- a/docs/my-website/docs/tutorials/lm_evaluation_harness.md +++ b/docs/my-website/docs/tutorials/lm_evaluation_harness.md @@ -117,7 +117,7 @@ Since LiteLLM provides an OpenAI compatible proxy `-t` and `-m` don't need to ch `-m` will remain gpt-3.5 ```shell -./fasteval -b human-eval-plus -t openai -m gpt-3.5-turbo +./fasteval -b human-eval-plus -t openai -m gpt-4o ``` ## FLASK - Fine-grained Language Model Evaluation diff --git a/docs/my-website/docs/tutorials/mock_completion.md b/docs/my-website/docs/tutorials/mock_completion.md index cadd65e46dc..63920b3a983 100644 --- a/docs/my-website/docs/tutorials/mock_completion.md +++ b/docs/my-website/docs/tutorials/mock_completion.md @@ -8,7 +8,7 @@ Pass `mock_response` to `litellm.completion` and litellm will directly return th ```python from litellm import completion -model = "gpt-3.5-turbo" +model = "gpt-4o" messages = [{"role":"user", "content":"Why is LiteLLM amazing?"}] completion(model=model, messages=messages, mock_response="It's simple to use and easy to get started") @@ -23,7 +23,7 @@ import pytest def test_completion_openai(): try: response = completion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{"role":"user", "content":"Why is LiteLLM amazing?"}], mock_response="LiteLLM is awesome" ) diff --git a/docs/my-website/docs/tutorials/model_fallbacks.md b/docs/my-website/docs/tutorials/model_fallbacks.md index def76e47329..275e7e9bc86 100644 --- a/docs/my-website/docs/tutorials/model_fallbacks.md +++ b/docs/my-website/docs/tutorials/model_fallbacks.md @@ -19,7 +19,7 @@ os.environ["AZURE_API_KEY"] = "" os.environ["AZURE_API_BASE"] = "" os.environ["AZURE_API_VERSION"] = "" -model_fallback_list = ["claude-instant-1", "gpt-3.5-turbo", "chatgpt-test"] +model_fallback_list = ["claude-instant-1", "gpt-4o", "chatgpt-test"] user_message = "Hello, how are you?" messages = [{ "content": user_message,"role": "user"}] @@ -50,7 +50,7 @@ os.environ["AZURE_API_KEY"] = "" os.environ["AZURE_API_BASE"] = "" os.environ["AZURE_API_VERSION"] = "" -context_window_fallback_list = [{"model":"gpt-3.5-turbo-16k", "max_tokens": 16385}, {"model":"gpt-4-32k", "max_tokens": 32768}, {"model": "claude-instant-1", "max_tokens":100000}] +context_window_fallback_list = [{"model":"gpt-4o-16k", "max_tokens": 16385}, {"model":"gpt-4-32k", "max_tokens": 32768}, {"model": "claude-instant-1", "max_tokens":100000}] user_message = "Hello, how are you?" messages = [{ "content": user_message,"role": "user"}] diff --git a/docs/my-website/docs/tutorials/msft_sso.md b/docs/my-website/docs/tutorials/msft_sso.md index 2936f27297f..4f6fa604fb4 100644 --- a/docs/my-website/docs/tutorials/msft_sso.md +++ b/docs/my-website/docs/tutorials/msft_sso.md @@ -126,7 +126,7 @@ litellm_settings: default_team_params: # Default Params to apply when litellm auto creates a team from SSO IDP provider max_budget: 100 # Optional[float], optional): $100 budget for the team budget_duration: 30d # Optional[str], optional): 30 days budget_duration for the team - models: ["gpt-3.5-turbo"] # Optional[List[str]], optional): models to be used by the team + models: ["gpt-4o"] # Optional[List[str]], optional): models to be used by the team ``` ### 3.2 Auto-create a new team on LiteLLM diff --git a/docs/my-website/docs/tutorials/presidio_pii_masking.md b/docs/my-website/docs/tutorials/presidio_pii_masking.md index d6fe1adbd01..9a22d6fdf48 100644 --- a/docs/my-website/docs/tutorials/presidio_pii_masking.md +++ b/docs/my-website/docs/tutorials/presidio_pii_masking.md @@ -113,9 +113,9 @@ Create a `config.yaml` file: ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: gpt-4o litellm_params: - model: openai/gpt-3.5-turbo + model: openai/gpt-4o api_key: os.environ/OPENAI_API_KEY guardrails: @@ -170,7 +170,7 @@ curl -X POST http://localhost:4000/chat/completions \ -H "Content-Type: application/json" \ -H "Authorization: Bearer sk-1234" \ -d '{ - "model": "gpt-3.5-turbo", + "model": "gpt-4o", "messages": [ { "role": "user", @@ -207,7 +207,7 @@ My name is , my email is , and my credit card is