diff --git a/docs/my-website/docs/completion/audio.md b/docs/my-website/docs/completion/audio.md index 96b5e4f41c6..bfdd3d90a13 100644 --- a/docs/my-website/docs/completion/audio.md +++ b/docs/my-website/docs/completion/audio.md @@ -216,8 +216,8 @@ Use `litellm.supports_audio_input(model="")` -> returns `True` if model can acce assert litellm.supports_audio_output(model="gpt-4o-audio-preview") == True assert litellm.supports_audio_input(model="gpt-4o-audio-preview") == True -assert litellm.supports_audio_output(model="gpt-3.5-turbo") == False -assert litellm.supports_audio_input(model="gpt-3.5-turbo") == False +assert litellm.supports_audio_output(model="gpt-4o") == False +assert litellm.supports_audio_input(model="gpt-4o") == False ``` diff --git a/docs/my-website/docs/completion/batching.md b/docs/my-website/docs/completion/batching.md index 5854f4db800..d588d7570fc 100644 --- a/docs/my-website/docs/completion/batching.md +++ b/docs/my-website/docs/completion/batching.md @@ -68,7 +68,7 @@ os.environ['OPENAI_API_KEY'] = "" os.environ['COHERE_API_KEY'] = "" response = batch_completion_models( - models=["gpt-3.5-turbo", "claude-instant-1.2", "command-nightly"], + models=["gpt-4o", "claude-instant-1.2", "command-nightly"], messages=[{"role": "user", "content": "Hey, how's it going"}] ) print(result) @@ -203,7 +203,7 @@ os.environ['OPENAI_API_KEY'] = "" os.environ['COHERE_API_KEY'] = "" responses = batch_completion_models_all_responses( - models=["gpt-3.5-turbo", "claude-instant-1.2", "command-nightly"], + models=["gpt-4o", "claude-instant-1.2", "command-nightly"], messages=[{"role": "user", "content": "Hey, how's it going"}] ) print(responses) @@ -259,7 +259,7 @@ print(responses) "id": "chatcmpl-80szFnKHzCxObW0RqCMw1hWW1Icrq", "object": "chat.completion", "created": 1695222061, - "model": "gpt-3.5-turbo-0613", + "model": "gpt-4o-0613", "choices": [ { "index": 0, diff --git a/docs/my-website/docs/completion/drop_params.md b/docs/my-website/docs/completion/drop_params.md index cc32d3bbd32..b22315d3f00 100644 --- a/docs/my-website/docs/completion/drop_params.md +++ b/docs/my-website/docs/completion/drop_params.md @@ -210,7 +210,7 @@ client = openai.OpenAI( ) response = client.chat.completions.create( - model="gpt-3.5-turbo", + model="gpt-4o", messages = [ { "role": "user", diff --git a/docs/my-website/docs/completion/function_call.md b/docs/my-website/docs/completion/function_call.md index f10df68bf6f..bd4864d4405 100644 --- a/docs/my-website/docs/completion/function_call.md +++ b/docs/my-website/docs/completion/function_call.md @@ -5,7 +5,7 @@ Use `litellm.supports_function_calling(model="")` -> returns `True` if model supports Function calling, `False` if not ```python -assert litellm.supports_function_calling(model="gpt-3.5-turbo") == True +assert litellm.supports_function_calling(model="gpt-4o") == True assert litellm.supports_function_calling(model="azure/gpt-4-1106-preview") == True assert litellm.supports_function_calling(model="palm/chat-bison") == False assert litellm.supports_function_calling(model="xai/grok-2-latest") == True @@ -24,7 +24,7 @@ assert litellm.supports_parallel_function_calling(model="gpt-4") == False ## Parallel Function calling Parallel function calling is the model's ability to perform multiple function calls together, allowing the effects and results of these function calls to be resolved in parallel -## Quick Start - gpt-3.5-turbo-1106 +## Quick Start - gpt-4o-1106 Open In Colab @@ -36,7 +36,7 @@ In this example we define a single function `get_current_weather`. - Step 3: Send the model the output from running the `get_current_weather` function -### Full Code - Parallel function calling with `gpt-3.5-turbo-1106` +### Full Code - Parallel function calling with `gpt-4o-1106` ```python import litellm @@ -84,7 +84,7 @@ def test_parallel_function_call(): } ] response = litellm.completion( - model="gpt-3.5-turbo-1106", + model="gpt-4o-1106", messages=messages, tools=tools, tool_choice="auto", # auto is default, but we'll be explicit @@ -122,7 +122,7 @@ def test_parallel_function_call(): } ) # extend conversation with function response second_response = litellm.completion( - model="gpt-3.5-turbo-1106", + model="gpt-4o-1106", messages=messages, ) # get a new response from the model where it can see the function response print("\nSecond LLM response:\n", second_response) @@ -134,7 +134,7 @@ test_parallel_function_call() ``` ### Explanation - Parallel function calling -Below is an explanation of what is happening in the code snippet above for Parallel function calling with `gpt-3.5-turbo-1106` +Below is an explanation of what is happening in the code snippet above for Parallel function calling with `gpt-4o-1106` ### Step1: litellm.completion() with `tools` set to `get_current_weather` ```python import litellm @@ -178,7 +178,7 @@ tools = [ ] response = litellm.completion( - model="gpt-3.5-turbo-1106", + model="gpt-4o-1106", messages=messages, tools=tools, tool_choice="auto", # auto is default, but we'll be explicit @@ -206,7 +206,7 @@ ModelResponse( ])) ], created=1700319953, - model='gpt-3.5-turbo-1106', + model='gpt-4o-1106', object='chat.completion', system_fingerprint='fp_eeff13170a', usage={'completion_tokens': 77, 'prompt_tokens': 88, 'total_tokens': 165}, @@ -255,7 +255,7 @@ if tool_calls: Once the functions are executed, send the model the information for each function call and its response. This allows the model to generate a new response considering the effects of the function calls. ```python second_response = litellm.completion( - model="gpt-3.5-turbo-1106", + model="gpt-4o-1106", messages=messages, ) print("Second Response\n", second_response) @@ -270,7 +270,7 @@ ModelResponse( message=Message(content="The current weather in San Francisco is 72°F, in Tokyo it's 10°C, and in Paris it's 22°C.", role='assistant')) ], created=1700319955, - model='gpt-3.5-turbo-1106', + model='gpt-4o-1106', object='chat.completion', system_fingerprint='fp_eeff13170a', usage={'completion_tokens': 28, 'prompt_tokens': 169, 'total_tokens': 197}, @@ -415,7 +415,7 @@ functions = [ } ] -response = completion(model="gpt-3.5-turbo-0613", messages=messages, functions=functions) +response = completion(model="gpt-4o-0613", messages=messages, functions=functions) print(response) ``` @@ -499,7 +499,7 @@ def get_current_weather(location: str, unit: str): functions = [litellm.utils.function_to_dict(get_current_weather)] -response = completion(model="gpt-3.5-turbo-0613", messages=messages, functions=functions) +response = completion(model="gpt-4o-0613", messages=messages, functions=functions) print(response) ``` diff --git a/docs/my-website/docs/completion/http_handler_config.md b/docs/my-website/docs/completion/http_handler_config.md index d4a25ce2043..bae5995d4a3 100644 --- a/docs/my-website/docs/completion/http_handler_config.md +++ b/docs/my-website/docs/completion/http_handler_config.md @@ -18,7 +18,7 @@ import litellm # Works exactly as before response = await litellm.acompletion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{"role": "user", "content": "Hello!"}] ) ``` @@ -39,7 +39,7 @@ session = aiohttp.ClientSession( litellm.base_llm_aiohttp_handler = BaseLLMAIOHTTPHandler(client_session=session) # All completions now use your session -response = await litellm.acompletion(model="gpt-3.5-turbo", messages=[...]) +response = await litellm.acompletion(model="gpt-4o", messages=[...]) ``` ## Common Patterns @@ -69,7 +69,7 @@ app = FastAPI(lifespan=lifespan) @app.post("/chat") async def chat(messages: list[dict]): - return await litellm.acompletion(model="gpt-3.5-turbo", messages=messages) + return await litellm.acompletion(model="gpt-4o", messages=messages) ``` ### Corporate Proxy diff --git a/docs/my-website/docs/completion/input.md b/docs/my-website/docs/completion/input.md index cc058935221..7e836078d5c 100644 --- a/docs/my-website/docs/completion/input.md +++ b/docs/my-website/docs/completion/input.md @@ -15,7 +15,7 @@ os.environ["OPENAI_API_KEY"] = "your-openai-key" ## SET MAX TOKENS - via completion() response = litellm.completion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{ "content": "Hello, how are you?","role": "user"}], max_tokens=10 ) diff --git a/docs/my-website/docs/completion/mock_requests.md b/docs/my-website/docs/completion/mock_requests.md index fc357b0d7d7..de2643dd4a8 100644 --- a/docs/my-website/docs/completion/mock_requests.md +++ b/docs/my-website/docs/completion/mock_requests.md @@ -8,7 +8,7 @@ This will return a response object with a default response (works for streaming ```python from litellm import completion -model = "gpt-3.5-turbo" +model = "gpt-4o" messages = [{"role":"user", "content":"This is a test request"}] completion(model=model, messages=messages, mock_response="It's simple to use and easy to get started") @@ -18,7 +18,7 @@ completion(model=model, messages=messages, mock_response="It's simple to use and ```python from litellm import completion -model = "gpt-3.5-turbo" +model = "gpt-4o" messages = [{"role": "user", "content": "Hey, I'm a mock request"}] response = completion(model=model, messages=messages, stream=True, mock_response="It's simple to use and easy to get started") for chunk in response: @@ -60,7 +60,7 @@ import pytest def test_completion_openai(): try: response = completion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{"role":"user", "content":"Why is LiteLLM amazing?"}], mock_response="LiteLLM is awesome" ) diff --git a/docs/my-website/docs/completion/model_alias.md b/docs/my-website/docs/completion/model_alias.md index 5fa83264993..d5620e510ef 100644 --- a/docs/my-website/docs/completion/model_alias.md +++ b/docs/my-website/docs/completion/model_alias.md @@ -1,6 +1,6 @@ # Model Alias -The model name you show an end-user might be different from the one you pass to LiteLLM - e.g. Displaying `GPT-3.5` while calling `gpt-3.5-turbo-16k` on the backend. +The model name you show an end-user might be different from the one you pass to LiteLLM - e.g. Displaying `GPT-3.5` while calling `gpt-4o-16k` on the backend. LiteLLM simplifies this by letting you pass in a model alias mapping. @@ -18,7 +18,7 @@ litellm.model_alias_map = { ### Relevant Code ```python model_alias_map = { - "GPT-3.5": "gpt-3.5-turbo-16k", + "GPT-3.5": "gpt-4o-16k", "llama2": "replicate/llama-2-70b-chat:2796ee9483c3fd7aa2e171d38f4ca12251a30609463dcfd4cd76703f22e96cdf" } @@ -37,7 +37,7 @@ os.environ["REPLICATE_API_KEY"] = "cohere key" ## set model alias map model_alias_map = { - "GPT-3.5": "gpt-3.5-turbo-16k", + "GPT-3.5": "gpt-4o-16k", "llama2": "replicate/llama-2-70b-chat:2796ee9483c3fd7aa2e171d38f4ca12251a30609463dcfd4cd76703f22e96cdf" } @@ -45,7 +45,7 @@ litellm.model_alias_map = model_alias_map messages = [{ "content": "Hello, how are you?","role": "user"}] -# call "gpt-3.5-turbo-16k" +# call "gpt-4o-16k" response = completion(model="GPT-3.5", messages=messages) # call replicate/llama-2-70b-chat:2796ee9483c3fd7aa2e171d38f4ca1... diff --git a/docs/my-website/docs/completion/prompt_caching.md b/docs/my-website/docs/completion/prompt_caching.md index 402c7b9f4c7..0d2ef499c08 100644 --- a/docs/my-website/docs/completion/prompt_caching.md +++ b/docs/my-website/docs/completion/prompt_caching.md @@ -700,7 +700,7 @@ response = client.chat.completions.with_raw_response.create( "role": "user", "content": "Say this is a test", }], - model="gpt-3.5-turbo", + model="gpt-4o", ) print(response.headers.get('x-litellm-response-cost')) diff --git a/docs/my-website/docs/completion/provider_specific_params.md b/docs/my-website/docs/completion/provider_specific_params.md index 250b410c9c4..92610984974 100644 --- a/docs/my-website/docs/completion/provider_specific_params.md +++ b/docs/my-website/docs/completion/provider_specific_params.md @@ -22,7 +22,7 @@ os.environ["OPENAI_API_KEY"] = "your-openai-key" ## SET MAX TOKENS - via completion() response_1 = litellm.completion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{ "content": "Hello, how are you?","role": "user"}], max_tokens=10 ) @@ -33,7 +33,7 @@ response_1_text = response_1.choices[0].message.content litellm.OpenAIConfig(max_tokens=10) response_2 = litellm.completion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{ "content": "Hello, how are you?","role": "user"}], ) diff --git a/docs/my-website/docs/completion/reliable_completions.md b/docs/my-website/docs/completion/reliable_completions.md index f38917fe53d..32b280e5886 100644 --- a/docs/my-website/docs/completion/reliable_completions.md +++ b/docs/my-website/docs/completion/reliable_completions.md @@ -25,7 +25,7 @@ messages = [{"content": user_message, "role": "user"}] # normal call response = completion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=messages, num_retries=2 ) @@ -43,10 +43,10 @@ response = completion( ```python from litellm import completion -fallback_dict = {"gpt-3.5-turbo": "gpt-3.5-turbo-16k"} +fallback_dict = {"gpt-4o": "gpt-4o-16k"} messages = [{"content": "how does a court case get to the Supreme Court?" * 500, "role": "user"}] -completion(model="gpt-3.5-turbo", messages=messages, context_window_fallback_dict=fallback_dict) +completion(model="gpt-4o", messages=messages, context_window_fallback_dict=fallback_dict) ``` ### Fallbacks - Switch Models/API Keys/API Bases (SDK) @@ -61,7 +61,7 @@ The `fallbacks` list should include the primary model you want to use, followed #### switch models ```python response = completion(model="bad-model", messages=messages, - fallbacks=["gpt-3.5-turbo" "command-nightly"]) + fallbacks=["gpt-4o" "command-nightly"]) ``` #### switch api keys/bases (E.g. azure deployment) @@ -84,12 +84,12 @@ Completion with 'bad-model': got exception Unable to map your input to a model. -completion call gpt-3.5-turbo +completion call gpt-4o { "id": "chatcmpl-7qTmVRuO3m3gIBg4aTmAumV1TmQhB", "object": "chat.completion", "created": 1692741891, - "model": "gpt-3.5-turbo-0613", + "model": "gpt-4o-0613", "choices": [ { "index": 0, diff --git a/docs/my-website/docs/completion/stream.md b/docs/my-website/docs/completion/stream.md index 088437a76d9..370f92e9c50 100644 --- a/docs/my-website/docs/completion/stream.md +++ b/docs/my-website/docs/completion/stream.md @@ -15,7 +15,7 @@ LiteLLM supports streaming the model response back by passing `stream=True` as a ```python from litellm import completion messages = [{"role": "user", "content": "Hey, how's it going?"}] -response = completion(model="gpt-3.5-turbo", messages=messages, stream=True) +response = completion(model="gpt-4o", messages=messages, stream=True) for part in response: print(part.choices[0].delta.content or "") ``` @@ -27,7 +27,7 @@ LiteLLM also exposes a helper function to rebuild the complete streaming respons ```python from litellm import completion messages = [{"role": "user", "content": "Hey, how's it going?"}] -response = completion(model="gpt-3.5-turbo", messages=messages, stream=True) +response = completion(model="gpt-4o", messages=messages, stream=True) for chunk in response: chunks.append(chunk) @@ -45,7 +45,7 @@ import asyncio async def test_get_response(): user_message = "Hello, how are you?" messages = [{"content": user_message, "role": "user"}] - response = await acompletion(model="gpt-3.5-turbo", messages=messages) + response = await acompletion(model="gpt-4o", messages=messages) return response response = asyncio.run(test_get_response()) @@ -66,7 +66,7 @@ async def completion_call(): try: print("test acompletion + streaming") response = await acompletion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{"content": "Hello, how are you?", "role": "user"}], stream=True ) @@ -106,7 +106,7 @@ chunks = [ "id": "chatcmpl-123", "object": "chat.completion.chunk", "created": 1694268190, - "model": "gpt-3.5-turbo-0125", + "model": "gpt-4o-0125", "system_fingerprint": "fp_44709d6fcb", "choices": [ {"index": 0, "delta": {"content": "How are you?"}, "finish_reason": "stop"} @@ -117,10 +117,10 @@ completion_stream = litellm.ModelResponseListIterator(model_responses=chunks) response = litellm.CustomStreamWrapper( completion_stream=completion_stream, - model="gpt-3.5-turbo", + model="gpt-4o", custom_llm_provider="cached_response", logging_obj=litellm.Logging( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{"role": "user", "content": "Hey"}], stream=True, call_type="completion", diff --git a/docs/my-website/docs/completion/token_usage.md b/docs/my-website/docs/completion/token_usage.md index d99564765a1..b5c617dd09f 100644 --- a/docs/my-website/docs/completion/token_usage.md +++ b/docs/my-website/docs/completion/token_usage.md @@ -7,7 +7,7 @@ LiteLLM returns `response_cost` in all calls. from litellm import completion response = litellm.completion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{"role": "user", "content": "Hey, how's it going?"}], mock_response="Hello world", ) @@ -49,7 +49,7 @@ from litellm import encode, decode sample_text = "Hellö World, this is my input string!" # openai encoding + decoding -openai_tokens = encode(model="gpt-3.5-turbo", text=sample_text) +openai_tokens = encode(model="gpt-4o", text=sample_text) print(openai_tokens) ``` @@ -62,8 +62,8 @@ from litellm import encode, decode sample_text = "Hellö World, this is my input string!" # openai encoding + decoding -openai_tokens = encode(model="gpt-3.5-turbo", text=sample_text) -openai_text = decode(model="gpt-3.5-turbo", tokens=openai_tokens) +openai_tokens = encode(model="gpt-4o", text=sample_text) +openai_text = decode(model="gpt-4o", tokens=openai_tokens) print(openai_text) ``` @@ -73,7 +73,7 @@ print(openai_text) from litellm import token_counter messages = [{"user": "role", "content": "Hey, how's it going"}] -print(token_counter(model="gpt-3.5-turbo", messages=messages)) +print(token_counter(model="gpt-4o", messages=messages)) ``` ### 4. `create_pretrained_tokenizer` and `create_tokenizer` @@ -100,7 +100,7 @@ from litellm import cost_per_token prompt_tokens = 5 completion_tokens = 10 -prompt_tokens_cost_usd_dollar, completion_tokens_cost_usd_dollar = cost_per_token(model="gpt-3.5-turbo", prompt_tokens=prompt_tokens, completion_tokens=completion_tokens) +prompt_tokens_cost_usd_dollar, completion_tokens_cost_usd_dollar = cost_per_token(model="gpt-4o", prompt_tokens=prompt_tokens, completion_tokens=completion_tokens) print(prompt_tokens_cost_usd_dollar, completion_tokens_cost_usd_dollar) ``` @@ -134,13 +134,13 @@ print(formatted_string) ``` ### 7. `get_max_tokens` -Input: Accepts a model name - e.g., gpt-3.5-turbo (to get a complete list, call litellm.model_list). +Input: Accepts a model name - e.g., gpt-4o (to get a complete list, call litellm.model_list). Output: Returns the maximum number of tokens allowed for the given model ```python from litellm import get_max_tokens -model = "gpt-3.5-turbo" +model = "gpt-4o" print(get_max_tokens(model)) # Output: 4097 ``` @@ -152,7 +152,7 @@ print(get_max_tokens(model)) # Output: 4097 ```python from litellm import model_cost -print(model_cost) # {'gpt-3.5-turbo': {'max_tokens': 4000, 'input_cost_per_token': 1.5e-06, 'output_cost_per_token': 2e-06}, ...} +print(model_cost) # {'gpt-4o': {'max_tokens': 4000, 'input_cost_per_token': 1.5e-06, 'output_cost_per_token': 2e-06}, ...} ``` ### 9. `register_model` diff --git a/docs/my-website/docs/completion/usage.md b/docs/my-website/docs/completion/usage.md index d610afeae55..fa66ae401ea 100644 --- a/docs/my-website/docs/completion/usage.md +++ b/docs/my-website/docs/completion/usage.md @@ -20,7 +20,7 @@ import os os.environ["OPENAI_API_KEY"] = "your-api-key" response = completion( - model="gpt-3.5-turbo", + model="gpt-4o", messages=[{ "content": "Hello, how are you?","role": "user"}] ) diff --git a/docs/my-website/docs/completion/vision.md b/docs/my-website/docs/completion/vision.md index 90d6b2393fb..c08aeff8531 100644 --- a/docs/my-website/docs/completion/vision.md +++ b/docs/my-website/docs/completion/vision.md @@ -120,7 +120,7 @@ Use `litellm.supports_vision(model="")` -> returns `True` if model supports `vis ```python assert litellm.supports_vision(model="openai/gpt-4-vision-preview") == True assert litellm.supports_vision(model="vertex_ai/gemini-1.0-pro-vision") == True -assert litellm.supports_vision(model="openai/gpt-3.5-turbo") == False +assert litellm.supports_vision(model="openai/gpt-4o") == False assert litellm.supports_vision(model="xai/grok-2-vision-latest") == True assert litellm.supports_vision(model="xai/grok-2-latest") == False ```