docs(proxy): replace gpt-3.5-turbo with gpt-4o in proxy docs

Co-authored-by: Krish Dholakia <krrishdholakia@gmail.com>
This commit is contained in:
Cursor Agent 2026-03-21 17:59:47 +00:00
parent 831bc1b6fd
commit ecab4713db
No known key found for this signature in database
10 changed files with 75 additions and 75 deletions

View file

@ -34,9 +34,9 @@ Caching can be enabled by adding the `cache` key in the `config.yaml`
```yaml
model_list:
- model_name: gpt-3.5-turbo
- model_name: gpt-4o
litellm_params:
model: gpt-3.5-turbo
model: gpt-4o
- model_name: text-embedding-ada-002
litellm_params:
model: text-embedding-ada-002
@ -382,9 +382,9 @@ one**
```yaml
model_list:
- model_name: gpt-3.5-turbo
- model_name: gpt-4o
litellm_params:
model: gpt-3.5-turbo
model: gpt-4o
- model_name: text-embedding-ada-002
litellm_params:
model: text-embedding-ada-002
@ -415,9 +415,9 @@ $ litellm --config /path/to/config.yaml
```yaml
model_list:
- model_name: gpt-3.5-turbo
- model_name: gpt-4o
litellm_params:
model: gpt-3.5-turbo
model: gpt-4o
- model_name: text-embedding-ada-002
litellm_params:
model: text-embedding-ada-002
@ -457,9 +457,9 @@ Caching can be enabled by adding the `cache` key in the `config.yaml`
```yaml
model_list:
- model_name: gpt-3.5-turbo
- model_name: gpt-4o
litellm_params:
model: gpt-3.5-turbo
model: gpt-4o
- model_name: azure-embedding-model
litellm_params:
model: azure/azure-embedding-model
@ -558,7 +558,7 @@ Send the same request twice:
curl http://0.0.0.0:4000/v1/chat/completions \
-H "Content-Type: application/json" \
-d '{
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"messages": [{"role": "user", "content": "write a poem about litellm!"}],
"temperature": 0.7
}'
@ -566,7 +566,7 @@ curl http://0.0.0.0:4000/v1/chat/completions \
curl http://0.0.0.0:4000/v1/chat/completions \
-H "Content-Type: application/json" \
-d '{
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"messages": [{"role": "user", "content": "write a poem about litellm!"}],
"temperature": 0.7
}'
@ -625,7 +625,7 @@ client = OpenAI(
chat_completion = client.chat.completions.create(
messages=[{"role": "user", "content": "Hello"}],
model="gpt-3.5-turbo",
model="gpt-4o",
extra_body={
"cache": {
"ttl": 300 # Cache response for 5 minutes
@ -643,7 +643,7 @@ curl http://localhost:4000/v1/chat/completions \
-H "Content-Type: application/json" \
-H "Authorization: Bearer sk-1234" \
-d '{
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"cache": {"ttl": 300},
"messages": [
{"role": "user", "content": "Hello"}
@ -671,7 +671,7 @@ client = OpenAI(
chat_completion = client.chat.completions.create(
messages=[{"role": "user", "content": "Hello"}],
model="gpt-3.5-turbo",
model="gpt-4o",
extra_body={
"cache": {
"s-maxage": 600 # Only use cache if less than 10 minutes old
@ -689,7 +689,7 @@ curl http://localhost:4000/v1/chat/completions \
-H "Content-Type: application/json" \
-H "Authorization: Bearer sk-1234" \
-d '{
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"cache": {"s-maxage": 600},
"messages": [
{"role": "user", "content": "Hello"}
@ -717,7 +717,7 @@ client = OpenAI(
chat_completion = client.chat.completions.create(
messages=[{"role": "user", "content": "Hello"}],
model="gpt-3.5-turbo",
model="gpt-4o",
extra_body={
"cache": {
"no-cache": True # Skip cache check, get fresh response
@ -735,7 +735,7 @@ curl http://localhost:4000/v1/chat/completions \
-H "Content-Type: application/json" \
-H "Authorization: Bearer sk-1234" \
-d '{
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"cache": {"no-cache": true},
"messages": [
{"role": "user", "content": "Hello"}
@ -763,7 +763,7 @@ client = OpenAI(
chat_completion = client.chat.completions.create(
messages=[{"role": "user", "content": "Hello"}],
model="gpt-3.5-turbo",
model="gpt-4o",
extra_body={
"cache": {
"no-store": True # Don't cache this response
@ -781,7 +781,7 @@ curl http://localhost:4000/v1/chat/completions \
-H "Content-Type: application/json" \
-H "Authorization: Bearer sk-1234" \
-d '{
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"cache": {"no-store": true},
"messages": [
{"role": "user", "content": "Hello"}
@ -809,7 +809,7 @@ client = OpenAI(
chat_completion = client.chat.completions.create(
messages=[{"role": "user", "content": "Hello"}],
model="gpt-3.5-turbo",
model="gpt-4o",
extra_body={
"cache": {
"namespace": "my-custom-namespace" # Store in custom namespace
@ -827,7 +827,7 @@ curl http://localhost:4000/v1/chat/completions \
-H "Content-Type: application/json" \
-H "Authorization: Bearer sk-1234" \
-d '{
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"cache": {"namespace": "my-custom-namespace"},
"messages": [
{"role": "user", "content": "Hello"}
@ -908,9 +908,9 @@ litellm_settings:
```yaml
model_list:
- model_name: gpt-3.5-turbo
- model_name: gpt-4o
litellm_params:
model: gpt-3.5-turbo
model: gpt-4o
- model_name: text-embedding-ada-002
litellm_params:
model: text-embedding-ada-002
@ -956,7 +956,7 @@ curl -i --location 'http://0.0.0.0:4000/chat/completions' \
--header 'Authorization: Bearer sk-1234' \
--header 'Content-Type: application/json' \
--data '{
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"user": "ishan",
"messages": [
{
@ -1029,7 +1029,7 @@ chat_completion = client.chat.completions.create(
"content": "Say this is a test",
}
],
model="gpt-3.5-turbo",
model="gpt-4o",
extra_body = { # OpenAI python accepts extra args in extra_body
"cache": {"use-cache": True}
}
@ -1045,7 +1045,7 @@ curl http://localhost:4000/v1/chat/completions \
-H "Content-Type: application/json" \
-H "Authorization: Bearer sk-1234" \
-d '{
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"cache": {"use-cache": True}
"messages": [
{"role": "user", "content": "Say this is a test"}

View file

@ -135,9 +135,9 @@ proxy_handler_instance = MyCustomHandler()
```yaml
model_list:
- model_name: gpt-3.5-turbo
- model_name: gpt-4o
litellm_params:
model: gpt-3.5-turbo
model: gpt-4o
litellm_settings:
callbacks: custom_callbacks.proxy_handler_instance # sets litellm.callbacks = [proxy_handler_instance]
@ -151,7 +151,7 @@ $ litellm /path/to/config.yaml
```shell
curl --location 'http://0.0.0.0:4000/chat/completions' \
--data ' {
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"messages": [
{
"role": "user",
@ -231,9 +231,9 @@ proxy_handler_instance = MyCustomHandler()
```yaml
model_list:
- model_name: gpt-3.5-turbo
- model_name: gpt-4o
litellm_params:
model: gpt-3.5-turbo
model: gpt-4o
litellm_settings:
callbacks: custom_callbacks.proxy_handler_instance # sets litellm.callbacks = [proxy_handler_instance]
@ -247,7 +247,7 @@ $ litellm /path/to/config.yaml
```shell
curl --location 'http://0.0.0.0:4000/chat/completions' \
--data ' {
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"messages": [
{
"role": "user",
@ -318,9 +318,9 @@ proxy_handler_instance = MyCustomHandler()
```yaml
model_list:
- model_name: gpt-3.5-turbo
- model_name: gpt-4o
litellm_params:
model: gpt-3.5-turbo
model: gpt-4o
litellm_settings:
callbacks: custom_callbacks.proxy_handler_instance # sets litellm.callbacks = [proxy_handler_instance]
@ -335,7 +335,7 @@ $ litellm /path/to/config.yaml
```shell
curl --location 'http://0.0.0.0:4000/chat/completions' \
--data ' {
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"messages": [
{
"role": "user",

View file

@ -41,7 +41,7 @@ user_config = {
{
'model_name': 'user-openai-instance',
'litellm_params': {
'model': 'gpt-3.5-turbo',
'model': 'gpt-4o',
'api_key': os.getenv('OPENAI_API_KEY'),
'timeout': 10,
},
@ -109,7 +109,7 @@ const userConfig = {
{
model_name: 'user-openai-instance',
litellm_params: {
model: 'gpt-3.5-turbo',
model: 'gpt-4o',
api_key: process.env.OPENAI_API_KEY,
timeout: 10,
},
@ -140,7 +140,7 @@ const openai = new OpenAI({
async function main() {
const chatCompletion = await openai.chat.completions.create({
messages: [{ role: 'user', content: 'Say this is a test' }],
model: 'gpt-3.5-turbo',
model: 'gpt-4o',
user_config: userConfig // # 👈 User config
});
}
@ -188,7 +188,7 @@ client = openai.OpenAI(
)
# request sent to model set on litellm proxy, `litellm --model`
response = client.chat.completions.create(model="gpt-3.5-turbo", messages = [
response = client.chat.completions.create(model="gpt-4o", messages = [
{
"role": "user",
"content": "this is a test request, write a short poem"
@ -213,7 +213,7 @@ client = openai.OpenAI(
)
# request sent to model set on litellm proxy, `litellm --model`
response = client.chat.completions.create(model="gpt-3.5-turbo", messages = [
response = client.chat.completions.create(model="gpt-4o", messages = [
{
"role": "user",
"content": "this is a test request, write a short poem"
@ -245,7 +245,7 @@ const openai = new OpenAI({
async function main() {
const chatCompletion = await openai.chat.completions.create({
messages: [{ role: 'user', content: 'Say this is a test' }],
model: 'gpt-3.5-turbo',
model: 'gpt-4o',
api_key: "my-bad-key" // 👈 User Key
});
}
@ -272,7 +272,7 @@ client = openai.OpenAI(
# request sent to model set on litellm proxy, `litellm --model`
response = client.chat.completions.create(
model="gpt-3.5-turbo",
model="gpt-4o",
messages = [
{
"role": "user",

View file

@ -34,8 +34,8 @@ litellm_settings:
# Fallbacks, reliability
default_fallbacks: ["claude-opus"] # set default_fallbacks, in case a specific model group is misconfigured / bad.
content_policy_fallbacks: [{ "gpt-3.5-turbo-small": ["claude-opus"] }] # fallbacks for ContentPolicyErrors
context_window_fallbacks: [{ "gpt-3.5-turbo-small": ["gpt-3.5-turbo-large", "claude-opus"] }] # fallbacks for ContextWindowExceededErrors
content_policy_fallbacks: [{ "gpt-4o-small": ["claude-opus"] }] # fallbacks for ContentPolicyErrors
context_window_fallbacks: [{ "gpt-4o-small": ["gpt-4o-large", "claude-opus"] }] # fallbacks for ContextWindowExceededErrors
# MCP Aliases - Map aliases to MCP server names for easier tool access
mcp_aliases: {
@ -349,7 +349,7 @@ router_settings:
| model_group_alias | dict | Model group alias mapping. E.g. `{"claude-3-haiku": "claude-3-haiku-20240229"}` |
| num_retries | int | Number of retries for a request. Defaults to 3. |
| default_fallbacks | Optional[List[str]] | Fallbacks to try if no model group-specific fallbacks are defined. |
| caching_groups | Optional[List[tuple]] | List of model groups for caching across model groups. Defaults to None. - e.g. caching_groups=[("openai-gpt-3.5-turbo", "azure-gpt-3.5-turbo")]|
| caching_groups | Optional[List[tuple]] | List of model groups for caching across model groups. Defaults to None. - e.g. caching_groups=[("openai-gpt-4o", "azure-gpt-4o")]|
| alerting_config | AlertingConfig | [SDK-only arg] Slack alerting configuration. Defaults to None. [Further Docs](../routing.md#alerting-) |
| assistants_config | AssistantsConfig | Set on proxy via `assistant_settings`. [Further docs](../assistants.md) |
| set_verbose | boolean | [DEPRECATED PARAM - see debug docs](./debugging) If true, sets the logging level to verbose. |

View file

@ -400,9 +400,9 @@ model_list:
model: gpt-4o
api_key: <my-openai-key>
rpm: 200
- model_name: gpt-3.5-turbo-16k
- model_name: gpt-4o-16k
litellm_params:
model: gpt-3.5-turbo-16k
model: gpt-4o-16k
api_key: <my-openai-key>
rpm: 100
@ -410,7 +410,7 @@ litellm_settings:
num_retries: 3 # retry call 3 times on each model_name (e.g. zephyr-beta)
request_timeout: 10 # raise Timeout error if call takes longer than 10s. Sets litellm.request_timeout
fallbacks: [{"zephyr-beta": ["gpt-4o"]}] # fallback to gpt-4o if call fails num_retries
context_window_fallbacks: [{"zephyr-beta": ["gpt-3.5-turbo-16k"]}, {"gpt-4o": ["gpt-3.5-turbo-16k"]}] # fallback to gpt-3.5-turbo-16k if context window error
context_window_fallbacks: [{"zephyr-beta": ["gpt-4o-16k"]}, {"gpt-4o": ["gpt-4o-16k"]}] # fallback to gpt-4o-16k if context window error
allowed_fails: 3 # cooldown model if it fails > 1 call in a minute.
router_settings: # router_settings are optional
@ -497,9 +497,9 @@ Supported Environments:
2. For each model set the list of supported environments in `model_info.supported_environments`
```yaml
model_list:
- model_name: gpt-3.5-turbo-16k
- model_name: gpt-4o-16k
litellm_params:
model: openai/gpt-3.5-turbo-16k
model: openai/gpt-4o-16k
api_key: os.environ/OPENAI_API_KEY
model_info:
supported_environments: ["development", "production", "staging"]

View file

@ -388,7 +388,7 @@ client = openai.OpenAI(
response = client.chat.completions.create(
model="gpt-3.5-turbo",
model="gpt-4o",
messages = [
{
"role": "user",
@ -420,7 +420,7 @@ async function runOpenAI() {
try {
const response = await client.chat.completions.create({
model: "gpt-3.5-turbo",
model: "gpt-4o",
messages: [
{
role: "user",
@ -452,7 +452,7 @@ Pass `metadata` as part of the request body
curl --location 'http://0.0.0.0:4000/chat/completions' \
--header 'Content-Type: application/json' \
--data '{
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"messages": [
{
"role": "user",
@ -477,7 +477,7 @@ from langchain.schema import HumanMessage, SystemMessage
chat = ChatOpenAI(
openai_api_base="http://0.0.0.0:4000",
model = "gpt-3.5-turbo",
model = "gpt-4o",
temperature=0.1,
extra_body={
"metadata": {
@ -574,7 +574,7 @@ curl -X GET 'http://localhost:4000/global/spend/report?start_date=2024-04-01&end
"api_key": "898c28.." # the hashed api key
},
{
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"spend": 0.0000825,
"total_tokens": 85,
"api_key": "84dc28.." # the hashed api key
@ -627,22 +627,22 @@ Output from script
# Date: 2024-05-11T00:00:00+00:00
# Team: local_test_team
# Total Spend: 0.003675099999999999
# Metadata: [{'model': 'gpt-3.5-turbo', 'spend': 0.003675099999999999, 'api_key': 'b94d5e0bc3a71a573917fe1335dc0c14728c7016337451af9714924ff3a729db', 'total_tokens': 3105}]
# Metadata: [{'model': 'gpt-4o', 'spend': 0.003675099999999999, 'api_key': 'b94d5e0bc3a71a573917fe1335dc0c14728c7016337451af9714924ff3a729db', 'total_tokens': 3105}]
# Date: 2024-05-13T00:00:00+00:00
# Team: Unassigned Team
# Total Spend: 3.4e-05
# Metadata: [{'model': 'gpt-3.5-turbo', 'spend': 3.4e-05, 'api_key': '9569d13c9777dba68096dea49b0b03e0aaf4d2b65d4030eda9e8a2733c3cd6e0', 'total_tokens': 50}]
# Metadata: [{'model': 'gpt-4o', 'spend': 3.4e-05, 'api_key': '9569d13c9777dba68096dea49b0b03e0aaf4d2b65d4030eda9e8a2733c3cd6e0', 'total_tokens': 50}]
# Date: 2024-05-13T00:00:00+00:00
# Team: central
# Total Spend: 0.000684
# Metadata: [{'model': 'gpt-3.5-turbo', 'spend': 0.000684, 'api_key': '0323facdf3af551594017b9ef162434a9b9a8ca1bbd9ccbd9d6ce173b1015605', 'total_tokens': 498}]
# Metadata: [{'model': 'gpt-4o', 'spend': 0.000684, 'api_key': '0323facdf3af551594017b9ef162434a9b9a8ca1bbd9ccbd9d6ce173b1015605', 'total_tokens': 498}]
# Date: 2024-05-13T00:00:00+00:00
# Team: local_test_team
# Total Spend: 0.0005715000000000001
# Metadata: [{'model': 'gpt-3.5-turbo', 'spend': 0.0005715000000000001, 'api_key': 'b94d5e0bc3a71a573917fe1335dc0c14728c7016337451af9714924ff3a729db', 'total_tokens': 423}]
# Metadata: [{'model': 'gpt-4o', 'spend': 0.0005715000000000001, 'api_key': 'b94d5e0bc3a71a573917fe1335dc0c14728c7016337451af9714924ff3a729db', 'total_tokens': 423}]
```
</TabItem>
@ -700,7 +700,7 @@ curl -X GET 'http://localhost:4000/global/spend/report?start_date=2024-04-01&end
"api_key": "898c28.." # the hashed api key
},
{
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"spend": 0.0000825,
"total_tokens": 85,
"api_key": "84dc28.." # the hashed api key
@ -778,7 +778,7 @@ curl -X GET 'http://localhost:4000/global/spend/report?start_date=2024-04-01&end
"total_output_tokens": 872.0,
"model_details": [
{
"model": "gpt-3.5-turbo-instruct",
"model": "gpt-4o-instruct",
"total_cost": 5.85e-05,
"total_input_tokens": 15,
"total_output_tokens": 18
@ -798,7 +798,7 @@ curl -X GET 'http://localhost:4000/global/spend/report?start_date=2024-04-01&end
"total_output_tokens": 27.0,
"model_details": [
{
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"total_cost": 5.2499999999999995e-05,
"total_input_tokens": 24,
"total_output_tokens": 27
@ -931,7 +931,7 @@ client = openai.OpenAI(
# request sent to model set on litellm proxy, `litellm --model`
response = client.chat.completions.create(
model="gpt-3.5-turbo",
model="gpt-4o",
messages = [
{
"role": "user",
@ -961,7 +961,7 @@ client = openai.OpenAI(
# Pass spend logs metadata via headers
response = client.chat.completions.create(
model="gpt-3.5-turbo",
model="gpt-4o",
messages = [
{
"role": "user",
@ -992,7 +992,7 @@ async function runOpenAI() {
try {
const response = await client.chat.completions.create({
model: 'gpt-3.5-turbo',
model: 'gpt-4o',
messages: [
{
role: 'user',
@ -1029,7 +1029,7 @@ async function runOpenAI() {
try {
const response = await client.chat.completions.create({
model: 'gpt-3.5-turbo',
model: 'gpt-4o',
messages: [
{
role: 'user',
@ -1062,7 +1062,7 @@ Pass `metadata` as part of the request body
curl --location 'http://0.0.0.0:4000/chat/completions' \
--header 'Content-Type: application/json' \
--data '{
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"messages": [
{
"role": "user",
@ -1089,7 +1089,7 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \
--header 'Authorization: Bearer sk-1234' \
--header 'x-litellm-spend-logs-metadata: {"user_id": "12345", "project_id": "proj_abc", "request_type": "chat_completion"}' \
--data '{
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"messages": [
{
"role": "user",
@ -1113,7 +1113,7 @@ from langchain.schema import HumanMessage, SystemMessage
chat = ChatOpenAI(
openai_api_base="http://0.0.0.0:4000",
model = "gpt-3.5-turbo",
model = "gpt-4o",
temperature=0.1,
extra_body={
"metadata": {

View file

@ -81,7 +81,7 @@ custom_ui_sso_sign_in_handler = MyCustomSSOLoginHandler()
model_list:
- model_name: "openai-model"
litellm_params:
model: "gpt-3.5-turbo"
model: "gpt-4o"
general_settings:
custom_ui_sso_sign_in_handler: custom_sso_handler.custom_ui_sso_sign_in_handler
@ -184,7 +184,7 @@ e.g. if they're both in the same dir - `./config.yaml` and `./custom_sso.py`, th
model_list:
- model_name: "openai-model"
litellm_params:
model: "gpt-3.5-turbo"
model: "gpt-4o"
general_settings:
custom_sso: custom_sso.custom_sso_handler

View file

@ -36,7 +36,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
--header 'Content-Type: application/json' \
--header 'Authorization: Bearer sk-1234' \
--data '{
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"user": "customer-123",
"messages": [
{
@ -64,7 +64,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
--header 'Authorization: Bearer sk-1234' \
--header 'x-litellm-customer-id: customer-123' \
--data '{
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"messages": [
{
"role": "user",

View file

@ -48,7 +48,7 @@ POST Request Sent from LiteLLM:
curl -X POST \
https://api.openai.com/v1/chat/completions \
-H 'content-type: application/json' -H 'Authorization: Bearer sk-qnWGUIW9****************************************' \
-d '{"model": "gpt-3.5-turbo", "messages": [{"role": "user", "content": "this is a test request, write a short poem"}]}'
-d '{"model": "gpt-4o", "messages": [{"role": "user", "content": "this is a test request, write a short poem"}]}'
```
## Debug single request
@ -81,7 +81,7 @@ https://exampleopenaiendpoint-production.up.railway.app/chat/completions \
20:14:06 - LiteLLM:WARNING: litellm_logging.py:1015 - RAW RESPONSE:
{"id":"chatcmpl-817fc08f0d6c451485d571dab39b26a1","object":"chat.completion","created":1677652288,"model":"gpt-3.5-turbo-0301","system_fingerprint":"fp_44709d6fcb","choices":[{"index":0,"message":{"role":"assistant","content":"\n\nHello there, how may I assist you today?"},"logprobs":null,"finish_reason":"stop"}],"usage":{"prompt_tokens":9,"completion_tokens":12,"total_tokens":21}}
{"id":"chatcmpl-817fc08f0d6c451485d571dab39b26a1","object":"chat.completion","created":1677652288,"model":"gpt-4o-0301","system_fingerprint":"fp_44709d6fcb","choices":[{"index":0,"message":{"role":"assistant","content":"\n\nHello there, how may I assist you today?"},"logprobs":null,"finish_reason":"stop"}],"usage":{"prompt_tokens":9,"completion_tokens":12,"total_tokens":21}}
INFO: 127.0.0.1:56155 - "POST /chat/completions HTTP/1.1" 200 OK

View file

@ -34,7 +34,7 @@ curl --location 'http://0.0.0.0:4000/v1/chat/completions' \
--header 'Content-Type: application/json' \
--header 'Authorization: Bearer sk-1234' \
--data '{
"model": "gpt-3.5-turbo",
"model": "gpt-4o",
"messages": [{"role":"user","content":"What llm are you?"}],
"temperature": 0.7,
"max_tokens": 10,