mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
docs(proxy): replace gpt-3.5-turbo with gpt-4o in proxy docs
Co-authored-by: Krish Dholakia <krrishdholakia@gmail.com>
This commit is contained in:
parent
831bc1b6fd
commit
ecab4713db
10 changed files with 75 additions and 75 deletions
|
|
@ -34,9 +34,9 @@ Caching can be enabled by adding the `cache` key in the `config.yaml`
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
model: gpt-4o
|
||||
- model_name: text-embedding-ada-002
|
||||
litellm_params:
|
||||
model: text-embedding-ada-002
|
||||
|
|
@ -382,9 +382,9 @@ one**
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
model: gpt-4o
|
||||
- model_name: text-embedding-ada-002
|
||||
litellm_params:
|
||||
model: text-embedding-ada-002
|
||||
|
|
@ -415,9 +415,9 @@ $ litellm --config /path/to/config.yaml
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
model: gpt-4o
|
||||
- model_name: text-embedding-ada-002
|
||||
litellm_params:
|
||||
model: text-embedding-ada-002
|
||||
|
|
@ -457,9 +457,9 @@ Caching can be enabled by adding the `cache` key in the `config.yaml`
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
model: gpt-4o
|
||||
- model_name: azure-embedding-model
|
||||
litellm_params:
|
||||
model: azure/azure-embedding-model
|
||||
|
|
@ -558,7 +558,7 @@ Send the same request twice:
|
|||
curl http://0.0.0.0:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [{"role": "user", "content": "write a poem about litellm!"}],
|
||||
"temperature": 0.7
|
||||
}'
|
||||
|
|
@ -566,7 +566,7 @@ curl http://0.0.0.0:4000/v1/chat/completions \
|
|||
curl http://0.0.0.0:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [{"role": "user", "content": "write a poem about litellm!"}],
|
||||
"temperature": 0.7
|
||||
}'
|
||||
|
|
@ -625,7 +625,7 @@ client = OpenAI(
|
|||
|
||||
chat_completion = client.chat.completions.create(
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
extra_body={
|
||||
"cache": {
|
||||
"ttl": 300 # Cache response for 5 minutes
|
||||
|
|
@ -643,7 +643,7 @@ curl http://localhost:4000/v1/chat/completions \
|
|||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"cache": {"ttl": 300},
|
||||
"messages": [
|
||||
{"role": "user", "content": "Hello"}
|
||||
|
|
@ -671,7 +671,7 @@ client = OpenAI(
|
|||
|
||||
chat_completion = client.chat.completions.create(
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
extra_body={
|
||||
"cache": {
|
||||
"s-maxage": 600 # Only use cache if less than 10 minutes old
|
||||
|
|
@ -689,7 +689,7 @@ curl http://localhost:4000/v1/chat/completions \
|
|||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"cache": {"s-maxage": 600},
|
||||
"messages": [
|
||||
{"role": "user", "content": "Hello"}
|
||||
|
|
@ -717,7 +717,7 @@ client = OpenAI(
|
|||
|
||||
chat_completion = client.chat.completions.create(
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
extra_body={
|
||||
"cache": {
|
||||
"no-cache": True # Skip cache check, get fresh response
|
||||
|
|
@ -735,7 +735,7 @@ curl http://localhost:4000/v1/chat/completions \
|
|||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"cache": {"no-cache": true},
|
||||
"messages": [
|
||||
{"role": "user", "content": "Hello"}
|
||||
|
|
@ -763,7 +763,7 @@ client = OpenAI(
|
|||
|
||||
chat_completion = client.chat.completions.create(
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
extra_body={
|
||||
"cache": {
|
||||
"no-store": True # Don't cache this response
|
||||
|
|
@ -781,7 +781,7 @@ curl http://localhost:4000/v1/chat/completions \
|
|||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"cache": {"no-store": true},
|
||||
"messages": [
|
||||
{"role": "user", "content": "Hello"}
|
||||
|
|
@ -809,7 +809,7 @@ client = OpenAI(
|
|||
|
||||
chat_completion = client.chat.completions.create(
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
extra_body={
|
||||
"cache": {
|
||||
"namespace": "my-custom-namespace" # Store in custom namespace
|
||||
|
|
@ -827,7 +827,7 @@ curl http://localhost:4000/v1/chat/completions \
|
|||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"cache": {"namespace": "my-custom-namespace"},
|
||||
"messages": [
|
||||
{"role": "user", "content": "Hello"}
|
||||
|
|
@ -908,9 +908,9 @@ litellm_settings:
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
model: gpt-4o
|
||||
- model_name: text-embedding-ada-002
|
||||
litellm_params:
|
||||
model: text-embedding-ada-002
|
||||
|
|
@ -956,7 +956,7 @@ curl -i --location 'http://0.0.0.0:4000/chat/completions' \
|
|||
--header 'Authorization: Bearer sk-1234' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"user": "ishan",
|
||||
"messages": [
|
||||
{
|
||||
|
|
@ -1029,7 +1029,7 @@ chat_completion = client.chat.completions.create(
|
|||
"content": "Say this is a test",
|
||||
}
|
||||
],
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
extra_body = { # OpenAI python accepts extra args in extra_body
|
||||
"cache": {"use-cache": True}
|
||||
}
|
||||
|
|
@ -1045,7 +1045,7 @@ curl http://localhost:4000/v1/chat/completions \
|
|||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"cache": {"use-cache": True}
|
||||
"messages": [
|
||||
{"role": "user", "content": "Say this is a test"}
|
||||
|
|
|
|||
|
|
@ -135,9 +135,9 @@ proxy_handler_instance = MyCustomHandler()
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
model: gpt-4o
|
||||
|
||||
litellm_settings:
|
||||
callbacks: custom_callbacks.proxy_handler_instance # sets litellm.callbacks = [proxy_handler_instance]
|
||||
|
|
@ -151,7 +151,7 @@ $ litellm /path/to/config.yaml
|
|||
```shell
|
||||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--data ' {
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -231,9 +231,9 @@ proxy_handler_instance = MyCustomHandler()
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
model: gpt-4o
|
||||
|
||||
litellm_settings:
|
||||
callbacks: custom_callbacks.proxy_handler_instance # sets litellm.callbacks = [proxy_handler_instance]
|
||||
|
|
@ -247,7 +247,7 @@ $ litellm /path/to/config.yaml
|
|||
```shell
|
||||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--data ' {
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -318,9 +318,9 @@ proxy_handler_instance = MyCustomHandler()
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
model: gpt-4o
|
||||
|
||||
litellm_settings:
|
||||
callbacks: custom_callbacks.proxy_handler_instance # sets litellm.callbacks = [proxy_handler_instance]
|
||||
|
|
@ -335,7 +335,7 @@ $ litellm /path/to/config.yaml
|
|||
```shell
|
||||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--data ' {
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
|
|||
|
|
@ -41,7 +41,7 @@ user_config = {
|
|||
{
|
||||
'model_name': 'user-openai-instance',
|
||||
'litellm_params': {
|
||||
'model': 'gpt-3.5-turbo',
|
||||
'model': 'gpt-4o',
|
||||
'api_key': os.getenv('OPENAI_API_KEY'),
|
||||
'timeout': 10,
|
||||
},
|
||||
|
|
@ -109,7 +109,7 @@ const userConfig = {
|
|||
{
|
||||
model_name: 'user-openai-instance',
|
||||
litellm_params: {
|
||||
model: 'gpt-3.5-turbo',
|
||||
model: 'gpt-4o',
|
||||
api_key: process.env.OPENAI_API_KEY,
|
||||
timeout: 10,
|
||||
},
|
||||
|
|
@ -140,7 +140,7 @@ const openai = new OpenAI({
|
|||
async function main() {
|
||||
const chatCompletion = await openai.chat.completions.create({
|
||||
messages: [{ role: 'user', content: 'Say this is a test' }],
|
||||
model: 'gpt-3.5-turbo',
|
||||
model: 'gpt-4o',
|
||||
user_config: userConfig // # 👈 User config
|
||||
});
|
||||
}
|
||||
|
|
@ -188,7 +188,7 @@ client = openai.OpenAI(
|
|||
)
|
||||
|
||||
# request sent to model set on litellm proxy, `litellm --model`
|
||||
response = client.chat.completions.create(model="gpt-3.5-turbo", messages = [
|
||||
response = client.chat.completions.create(model="gpt-4o", messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "this is a test request, write a short poem"
|
||||
|
|
@ -213,7 +213,7 @@ client = openai.OpenAI(
|
|||
)
|
||||
|
||||
# request sent to model set on litellm proxy, `litellm --model`
|
||||
response = client.chat.completions.create(model="gpt-3.5-turbo", messages = [
|
||||
response = client.chat.completions.create(model="gpt-4o", messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "this is a test request, write a short poem"
|
||||
|
|
@ -245,7 +245,7 @@ const openai = new OpenAI({
|
|||
async function main() {
|
||||
const chatCompletion = await openai.chat.completions.create({
|
||||
messages: [{ role: 'user', content: 'Say this is a test' }],
|
||||
model: 'gpt-3.5-turbo',
|
||||
model: 'gpt-4o',
|
||||
api_key: "my-bad-key" // 👈 User Key
|
||||
});
|
||||
}
|
||||
|
|
@ -272,7 +272,7 @@ client = openai.OpenAI(
|
|||
|
||||
# request sent to model set on litellm proxy, `litellm --model`
|
||||
response = client.chat.completions.create(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
|
|||
|
|
@ -34,8 +34,8 @@ litellm_settings:
|
|||
|
||||
# Fallbacks, reliability
|
||||
default_fallbacks: ["claude-opus"] # set default_fallbacks, in case a specific model group is misconfigured / bad.
|
||||
content_policy_fallbacks: [{ "gpt-3.5-turbo-small": ["claude-opus"] }] # fallbacks for ContentPolicyErrors
|
||||
context_window_fallbacks: [{ "gpt-3.5-turbo-small": ["gpt-3.5-turbo-large", "claude-opus"] }] # fallbacks for ContextWindowExceededErrors
|
||||
content_policy_fallbacks: [{ "gpt-4o-small": ["claude-opus"] }] # fallbacks for ContentPolicyErrors
|
||||
context_window_fallbacks: [{ "gpt-4o-small": ["gpt-4o-large", "claude-opus"] }] # fallbacks for ContextWindowExceededErrors
|
||||
|
||||
# MCP Aliases - Map aliases to MCP server names for easier tool access
|
||||
mcp_aliases: {
|
||||
|
|
@ -349,7 +349,7 @@ router_settings:
|
|||
| model_group_alias | dict | Model group alias mapping. E.g. `{"claude-3-haiku": "claude-3-haiku-20240229"}` |
|
||||
| num_retries | int | Number of retries for a request. Defaults to 3. |
|
||||
| default_fallbacks | Optional[List[str]] | Fallbacks to try if no model group-specific fallbacks are defined. |
|
||||
| caching_groups | Optional[List[tuple]] | List of model groups for caching across model groups. Defaults to None. - e.g. caching_groups=[("openai-gpt-3.5-turbo", "azure-gpt-3.5-turbo")]|
|
||||
| caching_groups | Optional[List[tuple]] | List of model groups for caching across model groups. Defaults to None. - e.g. caching_groups=[("openai-gpt-4o", "azure-gpt-4o")]|
|
||||
| alerting_config | AlertingConfig | [SDK-only arg] Slack alerting configuration. Defaults to None. [Further Docs](../routing.md#alerting-) |
|
||||
| assistants_config | AssistantsConfig | Set on proxy via `assistant_settings`. [Further docs](../assistants.md) |
|
||||
| set_verbose | boolean | [DEPRECATED PARAM - see debug docs](./debugging) If true, sets the logging level to verbose. |
|
||||
|
|
|
|||
|
|
@ -400,9 +400,9 @@ model_list:
|
|||
model: gpt-4o
|
||||
api_key: <my-openai-key>
|
||||
rpm: 200
|
||||
- model_name: gpt-3.5-turbo-16k
|
||||
- model_name: gpt-4o-16k
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo-16k
|
||||
model: gpt-4o-16k
|
||||
api_key: <my-openai-key>
|
||||
rpm: 100
|
||||
|
||||
|
|
@ -410,7 +410,7 @@ litellm_settings:
|
|||
num_retries: 3 # retry call 3 times on each model_name (e.g. zephyr-beta)
|
||||
request_timeout: 10 # raise Timeout error if call takes longer than 10s. Sets litellm.request_timeout
|
||||
fallbacks: [{"zephyr-beta": ["gpt-4o"]}] # fallback to gpt-4o if call fails num_retries
|
||||
context_window_fallbacks: [{"zephyr-beta": ["gpt-3.5-turbo-16k"]}, {"gpt-4o": ["gpt-3.5-turbo-16k"]}] # fallback to gpt-3.5-turbo-16k if context window error
|
||||
context_window_fallbacks: [{"zephyr-beta": ["gpt-4o-16k"]}, {"gpt-4o": ["gpt-4o-16k"]}] # fallback to gpt-4o-16k if context window error
|
||||
allowed_fails: 3 # cooldown model if it fails > 1 call in a minute.
|
||||
|
||||
router_settings: # router_settings are optional
|
||||
|
|
@ -497,9 +497,9 @@ Supported Environments:
|
|||
2. For each model set the list of supported environments in `model_info.supported_environments`
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo-16k
|
||||
- model_name: gpt-4o-16k
|
||||
litellm_params:
|
||||
model: openai/gpt-3.5-turbo-16k
|
||||
model: openai/gpt-4o-16k
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
model_info:
|
||||
supported_environments: ["development", "production", "staging"]
|
||||
|
|
|
|||
|
|
@ -388,7 +388,7 @@ client = openai.OpenAI(
|
|||
|
||||
|
||||
response = client.chat.completions.create(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -420,7 +420,7 @@ async function runOpenAI() {
|
|||
|
||||
try {
|
||||
const response = await client.chat.completions.create({
|
||||
model: "gpt-3.5-turbo",
|
||||
model: "gpt-4o",
|
||||
messages: [
|
||||
{
|
||||
role: "user",
|
||||
|
|
@ -452,7 +452,7 @@ Pass `metadata` as part of the request body
|
|||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -477,7 +477,7 @@ from langchain.schema import HumanMessage, SystemMessage
|
|||
|
||||
chat = ChatOpenAI(
|
||||
openai_api_base="http://0.0.0.0:4000",
|
||||
model = "gpt-3.5-turbo",
|
||||
model = "gpt-4o",
|
||||
temperature=0.1,
|
||||
extra_body={
|
||||
"metadata": {
|
||||
|
|
@ -574,7 +574,7 @@ curl -X GET 'http://localhost:4000/global/spend/report?start_date=2024-04-01&end
|
|||
"api_key": "898c28.." # the hashed api key
|
||||
},
|
||||
{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"spend": 0.0000825,
|
||||
"total_tokens": 85,
|
||||
"api_key": "84dc28.." # the hashed api key
|
||||
|
|
@ -627,22 +627,22 @@ Output from script
|
|||
# Date: 2024-05-11T00:00:00+00:00
|
||||
# Team: local_test_team
|
||||
# Total Spend: 0.003675099999999999
|
||||
# Metadata: [{'model': 'gpt-3.5-turbo', 'spend': 0.003675099999999999, 'api_key': 'b94d5e0bc3a71a573917fe1335dc0c14728c7016337451af9714924ff3a729db', 'total_tokens': 3105}]
|
||||
# Metadata: [{'model': 'gpt-4o', 'spend': 0.003675099999999999, 'api_key': 'b94d5e0bc3a71a573917fe1335dc0c14728c7016337451af9714924ff3a729db', 'total_tokens': 3105}]
|
||||
|
||||
# Date: 2024-05-13T00:00:00+00:00
|
||||
# Team: Unassigned Team
|
||||
# Total Spend: 3.4e-05
|
||||
# Metadata: [{'model': 'gpt-3.5-turbo', 'spend': 3.4e-05, 'api_key': '9569d13c9777dba68096dea49b0b03e0aaf4d2b65d4030eda9e8a2733c3cd6e0', 'total_tokens': 50}]
|
||||
# Metadata: [{'model': 'gpt-4o', 'spend': 3.4e-05, 'api_key': '9569d13c9777dba68096dea49b0b03e0aaf4d2b65d4030eda9e8a2733c3cd6e0', 'total_tokens': 50}]
|
||||
|
||||
# Date: 2024-05-13T00:00:00+00:00
|
||||
# Team: central
|
||||
# Total Spend: 0.000684
|
||||
# Metadata: [{'model': 'gpt-3.5-turbo', 'spend': 0.000684, 'api_key': '0323facdf3af551594017b9ef162434a9b9a8ca1bbd9ccbd9d6ce173b1015605', 'total_tokens': 498}]
|
||||
# Metadata: [{'model': 'gpt-4o', 'spend': 0.000684, 'api_key': '0323facdf3af551594017b9ef162434a9b9a8ca1bbd9ccbd9d6ce173b1015605', 'total_tokens': 498}]
|
||||
|
||||
# Date: 2024-05-13T00:00:00+00:00
|
||||
# Team: local_test_team
|
||||
# Total Spend: 0.0005715000000000001
|
||||
# Metadata: [{'model': 'gpt-3.5-turbo', 'spend': 0.0005715000000000001, 'api_key': 'b94d5e0bc3a71a573917fe1335dc0c14728c7016337451af9714924ff3a729db', 'total_tokens': 423}]
|
||||
# Metadata: [{'model': 'gpt-4o', 'spend': 0.0005715000000000001, 'api_key': 'b94d5e0bc3a71a573917fe1335dc0c14728c7016337451af9714924ff3a729db', 'total_tokens': 423}]
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
|
@ -700,7 +700,7 @@ curl -X GET 'http://localhost:4000/global/spend/report?start_date=2024-04-01&end
|
|||
"api_key": "898c28.." # the hashed api key
|
||||
},
|
||||
{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"spend": 0.0000825,
|
||||
"total_tokens": 85,
|
||||
"api_key": "84dc28.." # the hashed api key
|
||||
|
|
@ -778,7 +778,7 @@ curl -X GET 'http://localhost:4000/global/spend/report?start_date=2024-04-01&end
|
|||
"total_output_tokens": 872.0,
|
||||
"model_details": [
|
||||
{
|
||||
"model": "gpt-3.5-turbo-instruct",
|
||||
"model": "gpt-4o-instruct",
|
||||
"total_cost": 5.85e-05,
|
||||
"total_input_tokens": 15,
|
||||
"total_output_tokens": 18
|
||||
|
|
@ -798,7 +798,7 @@ curl -X GET 'http://localhost:4000/global/spend/report?start_date=2024-04-01&end
|
|||
"total_output_tokens": 27.0,
|
||||
"model_details": [
|
||||
{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"total_cost": 5.2499999999999995e-05,
|
||||
"total_input_tokens": 24,
|
||||
"total_output_tokens": 27
|
||||
|
|
@ -931,7 +931,7 @@ client = openai.OpenAI(
|
|||
|
||||
# request sent to model set on litellm proxy, `litellm --model`
|
||||
response = client.chat.completions.create(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -961,7 +961,7 @@ client = openai.OpenAI(
|
|||
|
||||
# Pass spend logs metadata via headers
|
||||
response = client.chat.completions.create(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-4o",
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -992,7 +992,7 @@ async function runOpenAI() {
|
|||
|
||||
try {
|
||||
const response = await client.chat.completions.create({
|
||||
model: 'gpt-3.5-turbo',
|
||||
model: 'gpt-4o',
|
||||
messages: [
|
||||
{
|
||||
role: 'user',
|
||||
|
|
@ -1029,7 +1029,7 @@ async function runOpenAI() {
|
|||
|
||||
try {
|
||||
const response = await client.chat.completions.create({
|
||||
model: 'gpt-3.5-turbo',
|
||||
model: 'gpt-4o',
|
||||
messages: [
|
||||
{
|
||||
role: 'user',
|
||||
|
|
@ -1062,7 +1062,7 @@ Pass `metadata` as part of the request body
|
|||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -1089,7 +1089,7 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \
|
|||
--header 'Authorization: Bearer sk-1234' \
|
||||
--header 'x-litellm-spend-logs-metadata: {"user_id": "12345", "project_id": "proj_abc", "request_type": "chat_completion"}' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -1113,7 +1113,7 @@ from langchain.schema import HumanMessage, SystemMessage
|
|||
|
||||
chat = ChatOpenAI(
|
||||
openai_api_base="http://0.0.0.0:4000",
|
||||
model = "gpt-3.5-turbo",
|
||||
model = "gpt-4o",
|
||||
temperature=0.1,
|
||||
extra_body={
|
||||
"metadata": {
|
||||
|
|
|
|||
|
|
@ -81,7 +81,7 @@ custom_ui_sso_sign_in_handler = MyCustomSSOLoginHandler()
|
|||
model_list:
|
||||
- model_name: "openai-model"
|
||||
litellm_params:
|
||||
model: "gpt-3.5-turbo"
|
||||
model: "gpt-4o"
|
||||
|
||||
general_settings:
|
||||
custom_ui_sso_sign_in_handler: custom_sso_handler.custom_ui_sso_sign_in_handler
|
||||
|
|
@ -184,7 +184,7 @@ e.g. if they're both in the same dir - `./config.yaml` and `./custom_sso.py`, th
|
|||
model_list:
|
||||
- model_name: "openai-model"
|
||||
litellm_params:
|
||||
model: "gpt-3.5-turbo"
|
||||
model: "gpt-4o"
|
||||
|
||||
general_settings:
|
||||
custom_sso: custom_sso.custom_sso_handler
|
||||
|
|
|
|||
|
|
@ -36,7 +36,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
|||
--header 'Content-Type: application/json' \
|
||||
--header 'Authorization: Bearer sk-1234' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"user": "customer-123",
|
||||
"messages": [
|
||||
{
|
||||
|
|
@ -64,7 +64,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
|||
--header 'Authorization: Bearer sk-1234' \
|
||||
--header 'x-litellm-customer-id: customer-123' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
|
|||
|
|
@ -48,7 +48,7 @@ POST Request Sent from LiteLLM:
|
|||
curl -X POST \
|
||||
https://api.openai.com/v1/chat/completions \
|
||||
-H 'content-type: application/json' -H 'Authorization: Bearer sk-qnWGUIW9****************************************' \
|
||||
-d '{"model": "gpt-3.5-turbo", "messages": [{"role": "user", "content": "this is a test request, write a short poem"}]}'
|
||||
-d '{"model": "gpt-4o", "messages": [{"role": "user", "content": "this is a test request, write a short poem"}]}'
|
||||
```
|
||||
|
||||
## Debug single request
|
||||
|
|
@ -81,7 +81,7 @@ https://exampleopenaiendpoint-production.up.railway.app/chat/completions \
|
|||
|
||||
|
||||
20:14:06 - LiteLLM:WARNING: litellm_logging.py:1015 - RAW RESPONSE:
|
||||
{"id":"chatcmpl-817fc08f0d6c451485d571dab39b26a1","object":"chat.completion","created":1677652288,"model":"gpt-3.5-turbo-0301","system_fingerprint":"fp_44709d6fcb","choices":[{"index":0,"message":{"role":"assistant","content":"\n\nHello there, how may I assist you today?"},"logprobs":null,"finish_reason":"stop"}],"usage":{"prompt_tokens":9,"completion_tokens":12,"total_tokens":21}}
|
||||
{"id":"chatcmpl-817fc08f0d6c451485d571dab39b26a1","object":"chat.completion","created":1677652288,"model":"gpt-4o-0301","system_fingerprint":"fp_44709d6fcb","choices":[{"index":0,"message":{"role":"assistant","content":"\n\nHello there, how may I assist you today?"},"logprobs":null,"finish_reason":"stop"}],"usage":{"prompt_tokens":9,"completion_tokens":12,"total_tokens":21}}
|
||||
|
||||
|
||||
INFO: 127.0.0.1:56155 - "POST /chat/completions HTTP/1.1" 200 OK
|
||||
|
|
|
|||
|
|
@ -34,7 +34,7 @@ curl --location 'http://0.0.0.0:4000/v1/chat/completions' \
|
|||
--header 'Content-Type: application/json' \
|
||||
--header 'Authorization: Bearer sk-1234' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"model": "gpt-4o",
|
||||
"messages": [{"role":"user","content":"What llm are you?"}],
|
||||
"temperature": 0.7,
|
||||
"max_tokens": 10,
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue