mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
docs(proxy): replace gpt-3.5-turbo with gpt-4o in proxy guides
Update example model names across virtual keys, users, keys, onboarding, reliability, projects, team routing, model management, logging, timeouts, provider budgets, service accounts, pass-through, tag routing, and token auth. Co-authored-by: Krish Dholakia <krrishdholakia@gmail.com>
This commit is contained in:
parent
95c07ebfb3
commit
bb741e73fc
15 changed files with 162 additions and 162 deletions
|
|
@ -34,7 +34,7 @@ curl -i -sSL --location 'http://0.0.0.0:4000/chat/completions' \
|
|||
--header 'Authorization: Bearer sk-1234' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [{"role": "user", "content": "what llm are you"}]
|
||||
}' | grep 'x-litellm'
|
||||
```
|
||||
|
|
@ -70,9 +70,9 @@ Set `litellm.turn_off_message_logging=True` This will prevent the messages and r
|
|||
**1. Setup config.yaml**
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
model: gpt-4o
|
||||
litellm_settings:
|
||||
success_callback: ["langfuse"]
|
||||
turn_off_message_logging: True # 👈 Key Change
|
||||
|
|
@ -83,7 +83,7 @@ litellm_settings:
|
|||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -116,9 +116,9 @@ Example config.yaml
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
model: gpt-4o
|
||||
```
|
||||
|
||||
**2. Setup per request header**
|
||||
|
|
@ -129,7 +129,7 @@ curl -L -X POST 'http://0.0.0.0:4000/v1/chat/completions' \
|
|||
-H 'Authorization: Bearer sk-zV5HlSIm8ihj1F9C_ZbB1g' \
|
||||
-H 'x-litellm-enable-message-redaction: true' \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo-testing",
|
||||
"model": "gpt-4o-testing",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -176,7 +176,7 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \
|
|||
--header 'Content-Type: application/json' \
|
||||
--header 'LiteLLM-Disable-Message-Redaction: true' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -209,7 +209,7 @@ curl -L -X POST 'http://0.0.0.0:4000/v1/chat/completions' \
|
|||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer <litellm-api-key>' \
|
||||
-d '{
|
||||
"model": "openai/gpt-3.5-turbo",
|
||||
"model": "openai/gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -238,7 +238,7 @@ client = openai.OpenAI(
|
|||
|
||||
# request sent to model set on litellm proxy, `litellm --model`
|
||||
response = client.chat.completions.create(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -358,9 +358,9 @@ pip install langfuse>=2.0.0
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
model: gpt-4o
|
||||
litellm_settings:
|
||||
success_callback: ["langfuse"]
|
||||
```
|
||||
|
|
@ -404,7 +404,7 @@ Pass `metadata` as part of the request body
|
|||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -434,7 +434,7 @@ client = openai.OpenAI(
|
|||
|
||||
# request sent to model set on litellm proxy, `litellm --model`
|
||||
response = client.chat.completions.create(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -468,7 +468,7 @@ from langchain.schema import HumanMessage, SystemMessage
|
|||
|
||||
chat = ChatOpenAI(
|
||||
openai_api_base="http://0.0.0.0:4000",
|
||||
model = "gpt-3.5-turbo",
|
||||
model = "gpt-4o",
|
||||
temperature=0.1,
|
||||
extra_body={
|
||||
"metadata": {
|
||||
|
|
@ -648,7 +648,7 @@ Pass `metadata` as part of the request body
|
|||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -675,7 +675,7 @@ client = openai.OpenAI(
|
|||
|
||||
# request sent to model set on litellm proxy, `litellm --model`
|
||||
response = client.chat.completions.create(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -706,7 +706,7 @@ from langchain.schema import HumanMessage, SystemMessage
|
|||
|
||||
chat = ChatOpenAI(
|
||||
openai_api_base="http://0.0.0.0:4000",
|
||||
model = "gpt-3.5-turbo",
|
||||
model = "gpt-4o",
|
||||
temperature=0.1,
|
||||
extra_body={
|
||||
"metadata": {
|
||||
|
|
@ -783,7 +783,7 @@ Test Request
|
|||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data ' {
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -863,7 +863,7 @@ Test Request
|
|||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data ' {
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -909,7 +909,7 @@ Test Request
|
|||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data ' {
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -956,7 +956,7 @@ Test Request
|
|||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data ' {
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -1005,7 +1005,7 @@ Test Request
|
|||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data ' {
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -1288,7 +1288,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
|||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "system",
|
||||
|
|
@ -1326,9 +1326,9 @@ AWS_REGION_NAME = ""
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
model: gpt-4o
|
||||
litellm_settings:
|
||||
success_callback: ["s3_v2"]
|
||||
s3_callback_params:
|
||||
|
|
@ -1755,9 +1755,9 @@ In the config below, we pass
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
model: gpt-4o
|
||||
|
||||
litellm_settings:
|
||||
callbacks: custom_callbacks.proxy_handler_instance # sets litellm.callbacks = [proxy_handler_instance]
|
||||
|
|
@ -1798,9 +1798,9 @@ custom_handler = MyCustomHandler()
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
model: gpt-4o
|
||||
|
||||
litellm_settings:
|
||||
callbacks: ["s3://litellm-proxy/custom_callbacks.custom_handler"]
|
||||
|
|
@ -1810,9 +1810,9 @@ litellm_settings:
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
model: gpt-4o
|
||||
|
||||
litellm_settings:
|
||||
callbacks: ["gcs://my-gcs-bucket/custom_callbacks.custom_handler"]
|
||||
|
|
@ -1898,7 +1898,7 @@ litellm --config proxy_config.yaml
|
|||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Authorization: Bearer sk-1234' \
|
||||
--data ' {
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -1914,13 +1914,13 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \
|
|||
|
||||
```shell
|
||||
On Success
|
||||
Model: gpt-3.5-turbo,
|
||||
Model: gpt-4o,
|
||||
Messages: [{'role': 'user', 'content': 'good morning good sir'}],
|
||||
User: ishaan-app,
|
||||
Usage: {'completion_tokens': 10, 'prompt_tokens': 11, 'total_tokens': 21},
|
||||
Cost: 3.65e-05,
|
||||
Response: {'id': 'chatcmpl-8S8avKJ1aVBg941y5xzGMSKrYCMvN', 'choices': [{'finish_reason': 'stop', 'index': 0, 'message': {'content': 'Good morning! How can I assist you today?', 'role': 'assistant'}}], 'created': 1701716913, 'model': 'gpt-3.5-turbo-0613', 'object': 'chat.completion', 'system_fingerprint': None, 'usage': {'completion_tokens': 10, 'prompt_tokens': 11, 'total_tokens': 21}}
|
||||
Proxy Metadata: {'user_api_key': None, 'headers': Headers({'host': '0.0.0.0:4000', 'user-agent': 'curl/7.88.1', 'accept': '*/*', 'authorization': 'Bearer sk-1234', 'content-length': '199', 'content-type': 'application/x-www-form-urlencoded'}), 'model_group': 'gpt-3.5-turbo', 'deployment': 'gpt-3.5-turbo-ModelID-gpt-3.5-turbo'}
|
||||
Response: {'id': 'chatcmpl-8S8avKJ1aVBg941y5xzGMSKrYCMvN', 'choices': [{'finish_reason': 'stop', 'index': 0, 'message': {'content': 'Good morning! How can I assist you today?', 'role': 'assistant'}}], 'created': 1701716913, 'model': 'gpt-4o-0613', 'object': 'chat.completion', 'system_fingerprint': None, 'usage': {'completion_tokens': 10, 'prompt_tokens': 11, 'total_tokens': 21}}
|
||||
Proxy Metadata: {'user_api_key': None, 'headers': Headers({'host': '0.0.0.0:4000', 'user-agent': 'curl/7.88.1', 'accept': '*/*', 'authorization': 'Bearer sk-1234', 'content-length': '199', 'content-type': 'application/x-www-form-urlencoded'}), 'model_group': 'gpt-4o', 'deployment': 'gpt-4o-ModelID-gpt-4o'}
|
||||
```
|
||||
|
||||
#### Logging Proxy Request Object, Header, Url
|
||||
|
|
@ -2404,9 +2404,9 @@ AWS_REGION_NAME = ""
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
model: gpt-4o
|
||||
litellm_settings:
|
||||
success_callback: ["dynamodb"]
|
||||
dynamodb_table_name: your-table-name
|
||||
|
|
@ -2458,13 +2458,13 @@ Your logs should be available on DynamoDB
|
|||
"S": "{}"
|
||||
},
|
||||
"model": {
|
||||
"S": "gpt-3.5-turbo"
|
||||
"S": "gpt-4o"
|
||||
},
|
||||
"modelParameters": {
|
||||
"S": "{'temperature': 0.7, 'max_tokens': 100, 'user': 'ishaan-2'}"
|
||||
},
|
||||
"response": {
|
||||
"S": "ModelResponse(id='chatcmpl-8W15J4480a3fAQ1yQaMgtsKJAicen', choices=[Choices(finish_reason='stop', index=0, message=Message(content='Great! What can I assist you with?', role='assistant'))], created=1702641357, model='gpt-3.5-turbo-0613', object='chat.completion', system_fingerprint=None, usage=Usage(completion_tokens=9, prompt_tokens=11, total_tokens=20))"
|
||||
"S": "ModelResponse(id='chatcmpl-8W15J4480a3fAQ1yQaMgtsKJAicen', choices=[Choices(finish_reason='stop', index=0, message=Message(content='Great! What can I assist you with?', role='assistant'))], created=1702641357, model='gpt-4o-0613', object='chat.completion', system_fingerprint=None, usage=Usage(completion_tokens=9, prompt_tokens=11, total_tokens=20))"
|
||||
},
|
||||
"startTime": {
|
||||
"S": "2023-12-15 17:25:56.047035"
|
||||
|
|
@ -2531,9 +2531,9 @@ export SENTRY_ENVIRONMENT="development" # Controls the Sentry Environment (defau
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
model: gpt-4o
|
||||
litellm_settings:
|
||||
# other settings
|
||||
failure_callback: ["sentry"]
|
||||
|
|
@ -2571,9 +2571,9 @@ ATHINA_API_KEY = "your-athina-api-key"
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
model: gpt-4o
|
||||
litellm_settings:
|
||||
success_callback: ["athina"]
|
||||
```
|
||||
|
|
@ -2592,7 +2592,7 @@ Test Request
|
|||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data ' {
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -2625,9 +2625,9 @@ AZURE_CONTENT_SAFETY_KEY = "<your-azure-content-safety-key>"
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
model: gpt-4o
|
||||
litellm_settings:
|
||||
callbacks: ["azure_content_safety"]
|
||||
azure_content_safety_params:
|
||||
|
|
@ -2649,7 +2649,7 @@ Test Request
|
|||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data ' {
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -2672,9 +2672,9 @@ You can customize the thresholds for each category by setting the `thresholds` i
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
model: gpt-4o
|
||||
litellm_settings:
|
||||
callbacks: ["azure_content_safety"]
|
||||
azure_content_safety_params:
|
||||
|
|
|
|||
|
|
@ -48,14 +48,14 @@ Add a new model to the proxy via the `/model/new` API, to add models without res
|
|||
curl -X POST "http://0.0.0.0:4000/model/new" \
|
||||
-H "accept: application/json" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{ "model_name": "azure-gpt-turbo", "litellm_params": {"model": "azure/gpt-3.5-turbo", "api_key": "os.environ/AZURE_API_KEY", "api_base": "my-azure-api-base"} }'
|
||||
-d '{ "model_name": "azure-gpt-turbo", "litellm_params": {"model": "azure/gpt-4o", "api_key": "os.environ/AZURE_API_KEY", "api_base": "my-azure-api-base"} }'
|
||||
```
|
||||
</TabItem>
|
||||
<TabItem value="Yaml">
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo ### RECEIVED MODEL NAME ### `openai.chat.completions.create(model="gpt-3.5-turbo",...)`
|
||||
- model_name: gpt-4o ### RECEIVED MODEL NAME ### `openai.chat.completions.create(model="gpt-4o",...)`
|
||||
litellm_params: # all params accepted by litellm.completion() - https://github.com/BerriAI/litellm/blob/9b46ec05b02d36d6e4fb5c32321e51e7f56e4a6e/litellm/types/router.py#L297
|
||||
model: azure/gpt-turbo-small-eu ### MODEL NAME sent to `litellm.completion()` ###
|
||||
api_base: https://my-endpoint-europe-berri-992.openai.azure.com/
|
||||
|
|
|
|||
|
|
@ -344,7 +344,7 @@ anthropic_adapter = AnthropicAdapter()
|
|||
model_list:
|
||||
- model_name: my-claude-endpoint
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
model: gpt-4o
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
general_settings:
|
||||
|
|
|
|||
|
|
@ -41,7 +41,7 @@ curl --location 'http://0.0.0.0:4000/project/new' \
|
|||
--data '{
|
||||
"project_alias": "flight-search-assistant",
|
||||
"team_id": "ad898803-c8a3-4f4a-976a-a3c372cffa45",
|
||||
"models": ["gpt-4", "gpt-3.5-turbo"],
|
||||
"models": ["gpt-4", "gpt-4o"],
|
||||
"max_budget": 100,
|
||||
"metadata": {
|
||||
"use_case_id": "SNOW-12345",
|
||||
|
|
@ -56,7 +56,7 @@ curl --location 'http://0.0.0.0:4000/project/new' \
|
|||
"project_id": "e402a141-725a-4437-bff5-d47459189716",
|
||||
"project_alias": "flight-search-assistant",
|
||||
"team_id": "ad898803-c8a3-4f4a-976a-a3c372cffa45",
|
||||
"models": ["gpt-4", "gpt-3.5-turbo"],
|
||||
"models": ["gpt-4", "gpt-4o"],
|
||||
"max_budget": 100,
|
||||
...
|
||||
}
|
||||
|
|
@ -69,7 +69,7 @@ curl 'http://0.0.0.0:4000/key/generate' \
|
|||
--header 'Authorization: Bearer sk-1234' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data-raw '{
|
||||
"models": ["gpt-3.5-turbo", "gpt-4"],
|
||||
"models": ["gpt-4o", "gpt-4"],
|
||||
"metadata": {"user": "ishaan@berri.ai"},
|
||||
"project_id": "e402a141-725a-4437-bff5-d47459189716"
|
||||
}' | jq
|
||||
|
|
@ -219,11 +219,11 @@ curl --location 'http://0.0.0.0:4000/project/info?project_id=project-abc' \
|
|||
"project_id": "project-abc",
|
||||
"project_alias": "flight-search-assistant",
|
||||
"team_id": "team-123",
|
||||
"models": ["gpt-4", "gpt-3.5-turbo"],
|
||||
"models": ["gpt-4", "gpt-4o"],
|
||||
"spend": 45.67,
|
||||
"model_spend": {
|
||||
"gpt-4": 42.30,
|
||||
"gpt-3.5-turbo": 3.37
|
||||
"gpt-4o": 3.37
|
||||
},
|
||||
"litellm_budget_table": {
|
||||
"budget_id": "budget-xyz",
|
||||
|
|
@ -300,17 +300,17 @@ curl --location 'http://0.0.0.0:4000/project/new' \
|
|||
--data '{
|
||||
"project_alias": "multi-model-project",
|
||||
"team_id": "team-123",
|
||||
"models": ["gpt-4", "gpt-3.5-turbo", "claude-3-sonnet"],
|
||||
"models": ["gpt-4", "gpt-4o", "claude-3-sonnet"],
|
||||
"max_budget": 500,
|
||||
"metadata": {
|
||||
"model_tpm_limit": {
|
||||
"gpt-4": 50000,
|
||||
"gpt-3.5-turbo": 200000,
|
||||
"gpt-4o": 200000,
|
||||
"claude-3-sonnet": 100000
|
||||
},
|
||||
"model_rpm_limit": {
|
||||
"gpt-4": 50,
|
||||
"gpt-3.5-turbo": 500,
|
||||
"gpt-4o": 500,
|
||||
"claude-3-sonnet": 100
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -17,9 +17,9 @@ Set provider budgets in your `proxy_config.yaml` file
|
|||
#### Proxy Config setup
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: openai/gpt-3.5-turbo
|
||||
model: openai/gpt-4o
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
router_settings:
|
||||
|
|
@ -366,9 +366,9 @@ If you are using a multi-instance setup, you will need to set the Redis host, po
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: openai/gpt-3.5-turbo
|
||||
model: openai/gpt-4o
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
router_settings:
|
||||
|
|
|
|||
|
|
@ -19,7 +19,7 @@ Fallbacks are typically done from one `model_name` to another `model_name`.
|
|||
Key change:
|
||||
|
||||
```python
|
||||
fallbacks=[{"gpt-3.5-turbo": ["gpt-4"]}]
|
||||
fallbacks=[{"gpt-4o": ["gpt-4"]}]
|
||||
```
|
||||
|
||||
<Tabs>
|
||||
|
|
@ -30,7 +30,7 @@ from litellm import Router
|
|||
router = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "gpt-3.5-turbo",
|
||||
"model_name": "gpt-4o",
|
||||
"litellm_params": {
|
||||
"model": "azure/<your-deployment-name>",
|
||||
"api_base": "<your-azure-endpoint>",
|
||||
|
|
@ -48,7 +48,7 @@ router = Router(
|
|||
}
|
||||
}
|
||||
],
|
||||
fallbacks=[{"gpt-3.5-turbo": ["gpt-4"]}] # 👈 KEY CHANGE
|
||||
fallbacks=[{"gpt-4o": ["gpt-4"]}] # 👈 KEY CHANGE
|
||||
)
|
||||
|
||||
```
|
||||
|
|
@ -59,7 +59,7 @@ router = Router(
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: azure/<your-deployment-name>
|
||||
api_base: <your-azure-endpoint>
|
||||
|
|
@ -73,7 +73,7 @@ model_list:
|
|||
rpm: 6
|
||||
|
||||
router_settings:
|
||||
fallbacks: [{"gpt-3.5-turbo": ["gpt-4"]}]
|
||||
fallbacks: [{"gpt-4o": ["gpt-4"]}]
|
||||
```
|
||||
|
||||
|
||||
|
|
@ -138,7 +138,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
|||
|
||||
### Explanation
|
||||
|
||||
Fallbacks are done in-order - ["gpt-3.5-turbo, "gpt-4", "gpt-4-32k"], will do 'gpt-3.5-turbo' first, then 'gpt-4', etc.
|
||||
Fallbacks are done in-order - ["gpt-4o, "gpt-4", "gpt-4-32k"], will do 'gpt-4o' first, then 'gpt-4', etc.
|
||||
|
||||
You can also set [`default_fallbacks`](#default-fallbacks), in case a specific model group is misconfigured / bad.
|
||||
|
||||
|
|
@ -154,10 +154,10 @@ Set fallbacks in the `.completion()` call for SDK and client-side for proxy.
|
|||
|
||||
In this request the following will occur:
|
||||
1. The request to `model="zephyr-beta"` will fail
|
||||
2. litellm proxy will loop through all the model_groups specified in `fallbacks=["gpt-3.5-turbo"]`
|
||||
3. The request to `model="gpt-3.5-turbo"` will succeed and the client making the request will get a response from gpt-3.5-turbo
|
||||
2. litellm proxy will loop through all the model_groups specified in `fallbacks=["gpt-4o"]`
|
||||
3. The request to `model="gpt-4o"` will succeed and the client making the request will get a response from gpt-4o
|
||||
|
||||
👉 Key Change: `"fallbacks": ["gpt-3.5-turbo"]`
|
||||
👉 Key Change: `"fallbacks": ["gpt-4o"]`
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
|
@ -168,7 +168,7 @@ from litellm import Router
|
|||
router = Router(model_list=[..]) # defined in Step 1.
|
||||
|
||||
resp = router.completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Hey, how's it going?"}],
|
||||
mock_testing_fallbacks=True, # 👈 trigger fallbacks
|
||||
fallbacks=[
|
||||
|
|
@ -204,7 +204,7 @@ response = client.chat.completions.create(
|
|||
}
|
||||
],
|
||||
extra_body={
|
||||
"fallbacks": ["gpt-3.5-turbo"]
|
||||
"fallbacks": ["gpt-4o"]
|
||||
}
|
||||
)
|
||||
|
||||
|
|
@ -225,7 +225,7 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \
|
|||
"content": "what llm are you"
|
||||
}
|
||||
],
|
||||
"fallbacks": ["gpt-3.5-turbo"]
|
||||
"fallbacks": ["gpt-4o"]
|
||||
}'
|
||||
```
|
||||
</TabItem>
|
||||
|
|
@ -247,7 +247,7 @@ chat = ChatOpenAI(
|
|||
openai_api_base="http://0.0.0.0:4000",
|
||||
model="zephyr-beta",
|
||||
extra_body={
|
||||
"fallbacks": ["gpt-3.5-turbo"]
|
||||
"fallbacks": ["gpt-4o"]
|
||||
}
|
||||
)
|
||||
|
||||
|
|
@ -296,7 +296,7 @@ from litellm import Router
|
|||
router = Router(model_list=[..]) # defined in Step 1.
|
||||
|
||||
resp = router.completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Hey, how's it going?"}],
|
||||
mock_testing_fallbacks=True, # 👈 trigger fallbacks
|
||||
fallbacks=[
|
||||
|
|
@ -350,7 +350,7 @@ curl -L -X POST 'http://0.0.0.0:4000/v1/chat/completions' \
|
|||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -551,7 +551,7 @@ To set fallbacks, just do:
|
|||
|
||||
```
|
||||
litellm_settings:
|
||||
fallbacks: [{"zephyr-beta": ["gpt-3.5-turbo"]}]
|
||||
fallbacks: [{"zephyr-beta": ["gpt-4o"]}]
|
||||
```
|
||||
|
||||
**Covers all errors (429, 500, etc.)**
|
||||
|
|
@ -571,19 +571,19 @@ model_list:
|
|||
litellm_params:
|
||||
model: huggingface/HuggingFaceH4/zephyr-7b-beta
|
||||
api_base: http://0.0.0.0:8003
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
model: gpt-4o
|
||||
api_key: <my-openai-key>
|
||||
- model_name: gpt-3.5-turbo-16k
|
||||
- model_name: gpt-4o-16k
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo-16k
|
||||
model: gpt-4o-16k
|
||||
api_key: <my-openai-key>
|
||||
|
||||
litellm_settings:
|
||||
num_retries: 3 # retry call 3 times on each model_name (e.g. zephyr-beta)
|
||||
request_timeout: 10 # raise Timeout error if call takes longer than 10s. Sets litellm.request_timeout
|
||||
fallbacks: [{"zephyr-beta": ["gpt-3.5-turbo"]}] # fallback to gpt-3.5-turbo if call fails num_retries
|
||||
fallbacks: [{"zephyr-beta": ["gpt-4o"]}] # fallback to gpt-4o if call fails num_retries
|
||||
allowed_fails: 3 # cooldown model if it fails > 1 call in a minute.
|
||||
cooldown_time: 30 # how long to cooldown model if fails/min > allowed_fails
|
||||
```
|
||||
|
|
@ -749,14 +749,14 @@ For azure deployments, set the base model. Pick the base model from [this list](
|
|||
<Tabs>
|
||||
<TabItem value="same-group" label="Same Group">
|
||||
|
||||
Filter older instances of a model (e.g. gpt-3.5-turbo) with smaller context windows
|
||||
Filter older instances of a model (e.g. gpt-4o) with smaller context windows
|
||||
|
||||
```yaml
|
||||
router_settings:
|
||||
enable_pre_call_checks: true # 1. Enable pre-call checks
|
||||
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: azure/chatgpt-v-2
|
||||
api_base: os.environ/AZURE_API_BASE
|
||||
|
|
@ -765,9 +765,9 @@ model_list:
|
|||
model_info:
|
||||
base_model: azure/gpt-4-1106-preview # 2. 👈 (azure-only) SET BASE MODEL
|
||||
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo-1106
|
||||
model: gpt-4o-1106
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
```
|
||||
|
||||
|
|
@ -792,7 +792,7 @@ text = "What is the meaning of 42?" * 5000
|
|||
|
||||
# request sent to model set on litellm proxy, `litellm --model`
|
||||
response = client.chat.completions.create(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages = [
|
||||
{"role": "system", "content": text},
|
||||
{"role": "user", "content": "Who was Alexander?"},
|
||||
|
|
@ -813,7 +813,7 @@ router_settings:
|
|||
enable_pre_call_checks: true # 1. Enable pre-call checks
|
||||
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo-small
|
||||
- model_name: gpt-4o-small
|
||||
litellm_params:
|
||||
model: azure/chatgpt-v-2
|
||||
api_base: os.environ/AZURE_API_BASE
|
||||
|
|
@ -822,9 +822,9 @@ model_list:
|
|||
model_info:
|
||||
base_model: azure/gpt-4-1106-preview # 2. 👈 (azure-only) SET BASE MODEL
|
||||
|
||||
- model_name: gpt-3.5-turbo-large
|
||||
- model_name: gpt-4o-large
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo-1106
|
||||
model: gpt-4o-1106
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
- model_name: claude-opus
|
||||
|
|
@ -833,7 +833,7 @@ model_list:
|
|||
api_key: os.environ/ANTHROPIC_API_KEY
|
||||
|
||||
litellm_settings:
|
||||
context_window_fallbacks: [{"gpt-3.5-turbo-small": ["gpt-3.5-turbo-large", "claude-opus"]}]
|
||||
context_window_fallbacks: [{"gpt-4o-small": ["gpt-4o-large", "claude-opus"]}]
|
||||
```
|
||||
|
||||
**2. Start proxy**
|
||||
|
|
@ -857,7 +857,7 @@ text = "What is the meaning of 42?" * 5000
|
|||
|
||||
# request sent to model set on litellm proxy, `litellm --model`
|
||||
response = client.chat.completions.create(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages = [
|
||||
{"role": "system", "content": text},
|
||||
{"role": "user", "content": "Who was Alexander?"},
|
||||
|
|
@ -877,7 +877,7 @@ Fallback across providers (e.g. from Azure OpenAI to Anthropic) if you hit conte
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo-small
|
||||
- model_name: gpt-4o-small
|
||||
litellm_params:
|
||||
model: azure/chatgpt-v-2
|
||||
api_base: os.environ/AZURE_API_BASE
|
||||
|
|
@ -890,7 +890,7 @@ model_list:
|
|||
api_key: os.environ/ANTHROPIC_API_KEY
|
||||
|
||||
litellm_settings:
|
||||
content_policy_fallbacks: [{"gpt-3.5-turbo-small": ["claude-opus"]}]
|
||||
content_policy_fallbacks: [{"gpt-4o-small": ["claude-opus"]}]
|
||||
```
|
||||
|
||||
|
||||
|
|
@ -902,7 +902,7 @@ You can also set default_fallbacks, in case a specific model group is misconfigu
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo-small
|
||||
- model_name: gpt-4o-small
|
||||
litellm_params:
|
||||
model: azure/chatgpt-v-2
|
||||
api_base: os.environ/AZURE_API_BASE
|
||||
|
|
@ -920,7 +920,7 @@ litellm_settings:
|
|||
|
||||
This will default to claude-opus in case any model fails.
|
||||
|
||||
A model-specific fallbacks (e.g. `{"gpt-3.5-turbo-small": ["claude-opus"]}`) overrides default fallback.
|
||||
A model-specific fallbacks (e.g. `{"gpt-4o-small": ["claude-opus"]}`) overrides default fallback.
|
||||
|
||||
### EU-Region Filtering (Pre-Call Checks)
|
||||
|
||||
|
|
@ -937,7 +937,7 @@ router_settings:
|
|||
enable_pre_call_checks: true # 1. Enable pre-call checks
|
||||
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: azure/chatgpt-v-2
|
||||
api_base: os.environ/AZURE_API_BASE
|
||||
|
|
@ -945,9 +945,9 @@ model_list:
|
|||
api_version: "2023-07-01-preview"
|
||||
region_name: "eu" # 👈 SET EU-REGION
|
||||
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo-1106
|
||||
model: gpt-4o-1106
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
- model_name: gemini-pro
|
||||
|
|
@ -976,7 +976,7 @@ client = openai.OpenAI(
|
|||
|
||||
# request sent to model set on litellm proxy, `litellm --model`
|
||||
response = client.chat.completions.with_raw_response.create(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages = [{"role": "user", "content": "Who was Alexander?"}]
|
||||
)
|
||||
|
||||
|
|
@ -1055,7 +1055,7 @@ curl -L -X POST 'http://0.0.0.0:4000/v1/chat/completions' \
|
|||
"content": "List 5 important events in the XIX century"
|
||||
}
|
||||
],
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"disable_fallbacks": true # 👈 DISABLE FALLBACKS
|
||||
}'
|
||||
```
|
||||
|
|
|
|||
|
|
@ -53,7 +53,7 @@ curl --location 'http://localhost:4000/chat/completions' \
|
|||
--header 'Authorization: Bearer <sk-your-service-account>' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -86,7 +86,7 @@ curl --location 'http://localhost:4000/chat/completions' \
|
|||
--header 'Authorization: Bearer <sk-your-service-account>' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -115,7 +115,7 @@ Expected Response
|
|||
}
|
||||
],
|
||||
"created": 1677652288,
|
||||
"model": "gpt-3.5-turbo-0125",
|
||||
"model": "gpt-4o-0125",
|
||||
"object": "chat.completion",
|
||||
"system_fingerprint": "fp_44709d6fcb",
|
||||
"usage": {
|
||||
|
|
|
|||
|
|
@ -85,7 +85,7 @@ Response
|
|||
}
|
||||
],
|
||||
"created": 1677652288,
|
||||
"model": "gpt-3.5-turbo-0125",
|
||||
"model": "gpt-4o-0125",
|
||||
"object": "chat.completion",
|
||||
"system_fingerprint": "fp_44709d6fcb",
|
||||
"usage": {
|
||||
|
|
|
|||
|
|
@ -16,13 +16,13 @@ Create a config.yaml with 2 model groups + connected postgres db
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo-eu # 👈 Model Group 1
|
||||
- model_name: gpt-4o-eu # 👈 Model Group 1
|
||||
litellm_params:
|
||||
model: azure/chatgpt-v-2
|
||||
api_base: os.environ/AZURE_API_BASE_EU
|
||||
api_key: os.environ/AZURE_API_KEY_EU
|
||||
api_version: "2023-07-01-preview"
|
||||
- model_name: gpt-3.5-turbo-worldwide # 👈 Model Group 2
|
||||
- model_name: gpt-4o-worldwide # 👈 Model Group 2
|
||||
litellm_params:
|
||||
model: azure/chatgpt-v-2
|
||||
api_base: os.environ/AZURE_API_BASE
|
||||
|
|
@ -48,7 +48,7 @@ curl --location 'http://0.0.0.0:4000/team/new' \
|
|||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"team_alias": "my-new-team_4",
|
||||
"model_aliases": {"gpt-3.5-turbo": "gpt-3.5-turbo-eu"}
|
||||
"model_aliases": {"gpt-4o": "gpt-4o-eu"}
|
||||
}'
|
||||
|
||||
# Returns team_id: my-team-id
|
||||
|
|
@ -72,7 +72,7 @@ curl --location 'http://0.0.0.0:4000/v1/chat/completions' \
|
|||
--header 'Content-Type: application/json' \
|
||||
--header 'Authorization: Bearer sk-A1L0C3Px2LJl53sF_kTF9A' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo", # 👈 MODEL
|
||||
"model": "gpt-4o", # 👈 MODEL
|
||||
"messages": [{"role": "system", "content": "You'\''re an expert at writing poems"}, {"role": "user", "content": "Write me a poem"}, {"role": "user", "content": "What'\''s your name?"}],
|
||||
"user": "usha"
|
||||
}'
|
||||
|
|
|
|||
|
|
@ -55,7 +55,7 @@ from litellm import Router
|
|||
import asyncio
|
||||
|
||||
model_list = [{
|
||||
"model_name": "gpt-3.5-turbo",
|
||||
"model_name": "gpt-4o",
|
||||
"litellm_params": {
|
||||
"model": "azure/chatgpt-v-2",
|
||||
"api_key": os.getenv("AZURE_API_KEY"),
|
||||
|
|
@ -70,7 +70,7 @@ model_list = [{
|
|||
router = Router(model_list=model_list, routing_strategy="least-busy")
|
||||
async def router_acompletion():
|
||||
response = await router.acompletion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Hey, how's it going?"}]
|
||||
)
|
||||
print(response)
|
||||
|
|
@ -84,7 +84,7 @@ asyncio.run(router_acompletion())
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: azure/gpt-turbo-small-eu
|
||||
api_base: https://my-endpoint-europe-berri-992.openai.azure.com/
|
||||
|
|
@ -92,7 +92,7 @@ model_list:
|
|||
timeout: 0.1 # timeout in (seconds)
|
||||
stream_timeout: 0.01 # timeout for stream requests (seconds)
|
||||
max_retries: 5
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: azure/gpt-turbo-small-ca
|
||||
api_base: https://my-endpoint-canada-berri992.openai.azure.com/
|
||||
|
|
@ -130,7 +130,7 @@ model_list = [{...}]
|
|||
router = Router(model_list=model_list)
|
||||
|
||||
response = router.completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "what color is red"}],
|
||||
timeout=1
|
||||
)
|
||||
|
|
@ -146,7 +146,7 @@ response = router.completion(
|
|||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data-raw '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{"role": "user", "content": "what color is red"}
|
||||
],
|
||||
|
|
@ -167,7 +167,7 @@ client = openai.OpenAI(
|
|||
)
|
||||
|
||||
response = client.chat.completions.create(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[
|
||||
{"role": "user", "content": "what color is red"}
|
||||
],
|
||||
|
|
|
|||
|
|
@ -864,9 +864,9 @@ model_list:
|
|||
litellm_params:
|
||||
model: anthropic/claude-3-5-sonnet
|
||||
api_key: os.environ/ANTHROPIC_API_KEY
|
||||
- model_name: gpt-3.5-turbo-testing
|
||||
- model_name: gpt-4o-testing
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
model: gpt-4o
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
general_settings:
|
||||
|
|
@ -878,7 +878,7 @@ general_settings:
|
|||
- scope: litellm.api.consumer
|
||||
models: ["anthropic-claude"]
|
||||
- scope: litellm.api.gpt_3_5_turbo
|
||||
models: ["gpt-3.5-turbo-testing"]
|
||||
models: ["gpt-4o-testing"]
|
||||
enforce_scope_based_access: true # 👈 enforce scope-based access control
|
||||
enforce_rbac: true # 👈 enforces only a Team/User/ProxyAdmin can access the proxy.
|
||||
```
|
||||
|
|
@ -905,7 +905,7 @@ curl -L -X POST 'http://0.0.0.0:4000/v1/chat/completions' \
|
|||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer eyJhbGci...' \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo-testing",
|
||||
"model": "gpt-4o-testing",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
|
|||
|
|
@ -67,7 +67,7 @@ client = openai.OpenAI(
|
|||
|
||||
# request sent to model set on litellm proxy, `litellm --model`
|
||||
response = client.chat.completions.create(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -105,7 +105,7 @@ client = openai.AzureOpenAI(
|
|||
|
||||
# request sent to model set on litellm proxy, `litellm --model`
|
||||
response = client.chat.completions.create(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -169,7 +169,7 @@ Pass `metadata` as part of the request body
|
|||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -201,7 +201,7 @@ os.environ["OPENAI_API_KEY"] = "anything"
|
|||
|
||||
chat = ChatOpenAI(
|
||||
openai_api_base="http://0.0.0.0:4000",
|
||||
model = "gpt-3.5-turbo",
|
||||
model = "gpt-4o",
|
||||
temperature=0.1,
|
||||
extra_body={
|
||||
"metadata": {
|
||||
|
|
@ -261,7 +261,7 @@ const openai = new OpenAI({
|
|||
async function main() {
|
||||
const chatCompletion = await openai.chat.completions.create({
|
||||
messages: [{ role: 'user', content: 'Say this is a test' }],
|
||||
model: 'gpt-3.5-turbo',
|
||||
model: 'gpt-4o',
|
||||
}, {"metadata": {
|
||||
"generation_name": "ishaan-generation-openaijs-client",
|
||||
"generation_id": "openaijs-client-gen-id22",
|
||||
|
|
@ -372,7 +372,7 @@ client = openai.OpenAI(
|
|||
)
|
||||
|
||||
response = client.chat.completions.create(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Hello!"}],
|
||||
extra_body={
|
||||
"metadata": {
|
||||
|
|
@ -414,7 +414,7 @@ response = chat.invoke([HumanMessage(content="Generate a blog post")])
|
|||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [{"role": "user", "content": "Hello!"}],
|
||||
"metadata": {
|
||||
"tags": ["api-test", "development"],
|
||||
|
|
@ -438,7 +438,7 @@ const openai = new OpenAI({
|
|||
async function main() {
|
||||
const response = await openai.chat.completions.create({
|
||||
messages: [{ role: 'user', content: 'Hello!' }],
|
||||
model: 'gpt-3.5-turbo',
|
||||
model: 'gpt-4o',
|
||||
metadata: {
|
||||
tags: ["javascript-client", "api-test"],
|
||||
trace_user_id: "js-user-789"
|
||||
|
|
@ -815,7 +815,7 @@ client = openai.OpenAI(
|
|||
)
|
||||
|
||||
# request sent to model set on litellm proxy, `litellm --model`
|
||||
response = client.chat.completions.create(model="gpt-3.5-turbo", messages = [
|
||||
response = client.chat.completions.create(model="gpt-4o", messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "this is a test request, write a short poem"
|
||||
|
|
@ -830,7 +830,7 @@ print(response)
|
|||
|
||||
#### Start the LiteLLM proxy
|
||||
```shell
|
||||
litellm --model gpt-3.5-turbo
|
||||
litellm --model gpt-4o
|
||||
|
||||
#INFO: Proxy running on http://0.0.0.0:4000
|
||||
```
|
||||
|
|
@ -972,11 +972,11 @@ Use this when you want to send 1 request to N Models
|
|||
|
||||
#### Expected Request Format
|
||||
|
||||
Pass model as a string of comma separated value of models. Example `"model"="llama3,gpt-3.5-turbo"`
|
||||
Pass model as a string of comma separated value of models. Example `"model"="llama3,gpt-4o"`
|
||||
|
||||
This same request will be sent to the following model groups on the [litellm proxy config.yaml](https://docs.litellm.ai/docs/proxy/configs)
|
||||
- `model_name="llama3"`
|
||||
- `model_name="gpt-3.5-turbo"`
|
||||
- `model_name="gpt-4o"`
|
||||
|
||||
<Tabs>
|
||||
|
||||
|
|
@ -989,7 +989,7 @@ import openai
|
|||
client = openai.OpenAI(api_key="sk-1234", base_url="http://0.0.0.0:4000")
|
||||
|
||||
response = client.chat.completions.create(
|
||||
model="gpt-3.5-turbo,llama3",
|
||||
model="gpt-4o,llama3",
|
||||
messages=[
|
||||
{"role": "user", "content": "this is a test request, write a short poem"}
|
||||
],
|
||||
|
|
@ -1022,7 +1022,7 @@ Get a list of responses when `model` is passed as a list
|
|||
)
|
||||
],
|
||||
created=1715462919,
|
||||
model='gpt-3.5-turbo-0125',
|
||||
model='gpt-4o-0125',
|
||||
object='chat.completion',
|
||||
system_fingerprint=None,
|
||||
usage=CompletionUsage(
|
||||
|
|
@ -1072,7 +1072,7 @@ curl --location 'http://localhost:4000/chat/completions' \
|
|||
--header 'Authorization: Bearer sk-1234' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"model": "llama3,gpt-3.5-turbo",
|
||||
"model": "llama3,gpt-4o",
|
||||
"max_tokens": 10,
|
||||
"user": "litellm2",
|
||||
"messages": [
|
||||
|
|
@ -1128,7 +1128,7 @@ Get a list of responses when `model` is passed as a list
|
|||
}
|
||||
],
|
||||
"created": 1715459877,
|
||||
"model": "gpt-3.5-turbo-0125",
|
||||
"model": "gpt-4o-0125",
|
||||
"object": "chat.completion",
|
||||
"system_fingerprint": null,
|
||||
"usage": {
|
||||
|
|
|
|||
|
|
@ -63,7 +63,7 @@ curl -X POST http://localhost:4000/v1/chat/completions \
|
|||
-H "Authorization: Bearer <your-api-key>" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [{"role": "user", "content": "Hello!"}]
|
||||
}'
|
||||
```
|
||||
|
|
|
|||
|
|
@ -51,7 +51,7 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \
|
|||
--header 'Autherization: Bearer sk-1234' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -704,8 +704,8 @@ curl --location 'http://0.0.0.0:4000/team/new' \
|
|||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"team_id": "my-prod-team",
|
||||
"model_rpm_limit": {"gpt-4": 100, "gpt-3.5-turbo": 200},
|
||||
"model_tpm_limit": {"gpt-4": 10000, "gpt-3.5-turbo": 20000}
|
||||
"model_rpm_limit": {"gpt-4": 100, "gpt-4o": 200},
|
||||
"model_tpm_limit": {"gpt-4": 10000, "gpt-4o": 20000}
|
||||
}'
|
||||
```
|
||||
|
||||
|
|
@ -717,8 +717,8 @@ curl --location 'http://0.0.0.0:4000/team/update' \
|
|||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"team_id": "my-prod-team",
|
||||
"model_rpm_limit": {"gpt-4": 100, "gpt-3.5-turbo": 200},
|
||||
"model_tpm_limit": {"gpt-4": 10000, "gpt-3.5-turbo": 20000}
|
||||
"model_rpm_limit": {"gpt-4": 100, "gpt-4o": 200},
|
||||
"model_tpm_limit": {"gpt-4": 10000, "gpt-4o": 20000}
|
||||
}'
|
||||
```
|
||||
|
||||
|
|
@ -733,8 +733,8 @@ curl --location 'http://0.0.0.0:4000/team/update' \
|
|||
--data '{
|
||||
"team_id": "my-prod-team",
|
||||
"metadata": {
|
||||
"model_rpm_limit": {"gpt-4": 100, "gpt-3.5-turbo": 200},
|
||||
"model_tpm_limit": {"gpt-4": 10000, "gpt-3.5-turbo": 20000}
|
||||
"model_rpm_limit": {"gpt-4": 100, "gpt-4o": 200},
|
||||
"model_tpm_limit": {"gpt-4": 10000, "gpt-4o": 20000}
|
||||
}
|
||||
}'
|
||||
```
|
||||
|
|
@ -948,9 +948,9 @@ This will NOT apply if a key has a team_id (team budgets will apply then). [Tell
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: "gpt-3.5-turbo"
|
||||
- model_name: "gpt-4o"
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
model: gpt-4o
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
litellm_settings:
|
||||
|
|
@ -983,7 +983,7 @@ curl -L -X POST 'http://0.0.0.0:4000/chat/completions' \
|
|||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-X53RdxnDhzamRwjKXR4IHg' \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [{"role": "user", "content": "Hey, how's it going?"}]
|
||||
}'
|
||||
```
|
||||
|
|
|
|||
|
|
@ -43,7 +43,7 @@ model_list:
|
|||
- model_name: gpt-4
|
||||
litellm_params:
|
||||
model: ollama/llama2
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: ollama/llama2
|
||||
|
||||
|
|
@ -64,7 +64,7 @@ litellm --config /path/to/config.yaml
|
|||
curl 'http://0.0.0.0:4000/key/generate' \
|
||||
--header 'Authorization: Bearer <your-master-key>' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data-raw '{"models": ["gpt-3.5-turbo", "gpt-4"], "metadata": {"user": "ishaan@berri.ai"}}'
|
||||
--data-raw '{"models": ["gpt-4o", "gpt-4"], "metadata": {"user": "ishaan@berri.ai"}}'
|
||||
```
|
||||
|
||||
## Spend Tracking
|
||||
|
|
@ -106,12 +106,12 @@ This is automatically updated (in USD) when calls are made to /completions, /cha
|
|||
"spend": 0.0001065, # 👈 SPEND
|
||||
"expires": "2023-11-24T23:19:11.131000Z",
|
||||
"models": [
|
||||
"gpt-3.5-turbo",
|
||||
"gpt-4o",
|
||||
"gpt-4",
|
||||
"claude-2"
|
||||
],
|
||||
"aliases": {
|
||||
"mistral-7b": "gpt-3.5-turbo"
|
||||
"mistral-7b": "gpt-4o"
|
||||
},
|
||||
"config": {}
|
||||
}
|
||||
|
|
@ -147,7 +147,7 @@ curl --location 'http://localhost:4000/user/new' \
|
|||
curl 'http://0.0.0.0:4000/key/generate' \
|
||||
--header 'Authorization: Bearer <your-master-key>' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data-raw '{"models": ["gpt-3.5-turbo", "gpt-4"], "user_id": "my-unique-id"}'
|
||||
--data-raw '{"models": ["gpt-4o", "gpt-4"], "user_id": "my-unique-id"}'
|
||||
```
|
||||
|
||||
Returns a key - `sk-...`.
|
||||
|
|
@ -200,7 +200,7 @@ curl --location 'http://localhost:4000/team/new' \
|
|||
curl 'http://0.0.0.0:4000/key/generate' \
|
||||
--header 'Authorization: Bearer <your-master-key>' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data-raw '{"models": ["gpt-3.5-turbo", "gpt-4"], "team_id": "my-unique-id"}'
|
||||
--data-raw '{"models": ["gpt-4o", "gpt-4"], "team_id": "my-unique-id"}'
|
||||
```
|
||||
|
||||
Returns a key - `sk-...`.
|
||||
|
|
@ -265,7 +265,7 @@ curl -X POST "https://0.0.0.0:4000/key/generate" \
|
|||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"models": ["my-free-tier"],
|
||||
"aliases": {"gpt-3.5-turbo": "my-free-tier"}, # 👈 KEY CHANGE
|
||||
"aliases": {"gpt-4o": "my-free-tier"}, # 👈 KEY CHANGE
|
||||
"duration": "30min"
|
||||
}'
|
||||
```
|
||||
|
|
@ -279,7 +279,7 @@ curl -X POST "https://0.0.0.0:4000/key/generate" \
|
|||
-H "Authorization: Bearer <user-key>" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -336,7 +336,7 @@ curl http://localhost:4000/v1/chat/completions \
|
|||
|
||||
Expect to see a successful response from the litellm proxy since the key passed in `X-Litellm-Key` is valid
|
||||
```shell
|
||||
{"id":"chatcmpl-f9b2b79a7c30477ab93cd0e717d1773e","choices":[{"finish_reason":"stop","index":0,"message":{"content":"\n\nHello there, how may I assist you today?","role":"assistant","tool_calls":null,"function_call":null}}],"created":1677652288,"model":"gpt-3.5-turbo-0125","object":"chat.completion","system_fingerprint":"fp_44709d6fcb","usage":{"completion_tokens":12,"prompt_tokens":9,"total_tokens":21}
|
||||
{"id":"chatcmpl-f9b2b79a7c30477ab93cd0e717d1773e","choices":[{"finish_reason":"stop","index":0,"message":{"content":"\n\nHello there, how may I assist you today?","role":"assistant","tool_calls":null,"function_call":null}}],"created":1677652288,"model":"gpt-4o-0125","object":"chat.completion","system_fingerprint":"fp_44709d6fcb","usage":{"completion_tokens":12,"prompt_tokens":9,"total_tokens":21}
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
|
@ -473,7 +473,7 @@ e.g. if they're both in the same dir - `./config.yaml` and `./custom_auth.py`, t
|
|||
model_list:
|
||||
- model_name: "openai-model"
|
||||
litellm_params:
|
||||
model: "gpt-3.5-turbo"
|
||||
model: "gpt-4o"
|
||||
|
||||
litellm_settings:
|
||||
drop_params: True
|
||||
|
|
@ -548,7 +548,7 @@ curl 'http://localhost:4000/key/sk-1234/regenerate' \
|
|||
},
|
||||
"models": [
|
||||
"gpt-4",
|
||||
"gpt-3.5-turbo"
|
||||
"gpt-4o"
|
||||
],
|
||||
"grace_period": "48h"
|
||||
}'
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue