diff --git a/docs/my-website/docs/proxy/caching.md b/docs/my-website/docs/proxy/caching.md
index 6da977c8b05..87e6a6fdb6e 100644
--- a/docs/my-website/docs/proxy/caching.md
+++ b/docs/my-website/docs/proxy/caching.md
@@ -1,28 +1,29 @@
-import Tabs from '@theme/Tabs';
-import TabItem from '@theme/TabItem';
+import Tabs from '@theme/Tabs'; import TabItem from '@theme/TabItem';
-# Caching
+# Caching
-:::note
+:::note
For OpenAI/Anthropic Prompt Caching, go [here](../completion/prompt_caching.md)
:::
-Cache LLM Responses. LiteLLM's caching system stores and reuses LLM responses to save costs and reduce latency. When you make the same request twice, the cached response is returned instead of calling the LLM API again.
-
-
+Cache LLM Responses. LiteLLM's caching system stores and reuses LLM responses to save costs and
+reduce latency. When you make the same request twice, the cached response is returned instead of
+calling the LLM API again.
### Supported Caches
- In Memory Cache
- Disk Cache
-- Redis Cache
+- Redis Cache
- Qdrant Semantic Cache
- Redis Semantic Cache
-- s3 Bucket Cache
+- S3 Bucket Cache
+- GCS Bucket Cache
## Quick Start
+
@@ -30,6 +31,7 @@ Cache LLM Responses. LiteLLM's caching system stores and reuses LLM responses to
Caching can be enabled by adding the `cache` key in the `config.yaml`
#### Step 1: Add `cache` to the config.yaml
+
```yaml
model_list:
- model_name: gpt-3.5-turbo
@@ -41,18 +43,19 @@ model_list:
litellm_settings:
set_verbose: True
- cache: True # set cache responses to True, litellm defaults to using a redis cache
+ cache: True # set cache responses to True, litellm defaults to using a redis cache
```
-#### [OPTIONAL] Step 1.5: Add redis namespaces, default ttl
+#### [OPTIONAL] Step 1.5: Add redis namespaces, default ttl
#### Namespace
+
If you want to create some folder for your keys, you can set a namespace, like this:
```yaml
litellm_settings:
- cache: true
- cache_params: # set cache params for redis
+ cache: true
+ cache_params: # set cache params for redis
type: redis
namespace: "litellm.caching.caching"
```
@@ -63,7 +66,7 @@ and keys will be stored like:
litellm.caching.caching:
```
-#### Redis Cluster
+#### Redis Cluster
@@ -75,12 +78,11 @@ model_list:
litellm_params:
model: "*"
-
litellm_settings:
cache: True
cache_params:
type: redis
- redis_startup_nodes: [{"host": "127.0.0.1", "port": "7001"}]
+ redis_startup_nodes: [{ "host": "127.0.0.1", "port": "7001" }]
```
@@ -121,8 +123,7 @@ print("REDIS_CLUSTER_NODES", os.environ["REDIS_CLUSTER_NODES"])
-#### Redis Sentinel
-
+#### Redis Sentinel
@@ -134,7 +135,6 @@ model_list:
litellm_params:
model: "*"
-
litellm_settings:
cache: true
cache_params:
@@ -181,18 +181,17 @@ print("REDIS_SENTINEL_NODES", os.environ["REDIS_SENTINEL_NODES"])
```yaml
litellm_settings:
- cache: true
- cache_params: # set cache params for redis
+ cache: true
+ cache_params: # set cache params for redis
type: redis
ttl: 600 # will be cached on redis for 600s
- # default_in_memory_ttl: Optional[float], default is None. time in seconds.
- # default_in_redis_ttl: Optional[float], default is None. time in seconds.
+ # default_in_memory_ttl: Optional[float], default is None. time in seconds.
+ # default_in_redis_ttl: Optional[float], default is None. time in seconds.
```
-
#### SSL
-just set `REDIS_SSL="True"` in your .env, and LiteLLM will pick this up.
+just set `REDIS_SSL="True"` in your .env, and LiteLLM will pick this up.
```env
REDIS_SSL="True"
@@ -204,14 +203,14 @@ For quick testing, you can also use REDIS_URL, eg.:
REDIS_URL="rediss://.."
```
-but we **don't** recommend using REDIS_URL in prod. We've noticed a performance difference between using it vs. redis_host, port, etc.
+but we **don't** recommend using REDIS_URL in prod. We've noticed a performance difference between
+using it vs. redis_host, port, etc.
#### GCP IAM Authentication
For GCP Memorystore Redis with IAM authentication, install the required dependency:
-:::info
-IAM authentication for redis is only supported via GCP and only on Redis Clusters for now.
+:::info IAM authentication for redis is only supported via GCP and only on Redis Clusters for now.
:::
```shell
@@ -229,7 +228,8 @@ litellm_settings:
cache: True
cache_params:
type: redis
- redis_startup_nodes: [{"host": "10.128.0.2", "port": 6379}, {"host": "10.128.0.2", "port": 11008}]
+ redis_startup_nodes:
+ [{ "host": "10.128.0.2", "port": 6379 }, { "host": "10.128.0.2", "port": 11008 }]
gcp_service_account: "projects/-/serviceAccounts/your-sa@project.iam.gserviceaccount.com"
ssl: true
ssl_cert_reqs: null
@@ -242,7 +242,6 @@ litellm_settings:
You can configure GCP IAM Redis authentication in your .env:
-
For Redis Cluster:
```env
@@ -283,24 +282,29 @@ Set either `REDIS_URL` or the `REDIS_HOST` in your os environment, to enable cac
```
**Additional kwargs**
-You can pass in any additional redis.Redis arg, by storing the variable + value in your os environment, like this:
+You can pass in any additional redis.Redis arg, by storing the variable + value in your os
+environment, like this:
+
```shell
REDIS_ = ""
-```
+```
[**See how it's read from the environment**](https://github.com/BerriAI/litellm/blob/4d7ff1b33b9991dcf38d821266290631d9bcd2dd/litellm/_redis.py#L40)
+
#### Step 3: Run proxy with config
+
```shell
$ litellm --config /path/to/config.yaml
```
-
+
Caching can be enabled by adding the `cache` key in the `config.yaml`
#### Step 1: Add `cache` to the config.yaml
+
```yaml
model_list:
- model_name: fake-openai-endpoint
@@ -315,13 +319,13 @@ model_list:
litellm_settings:
set_verbose: True
- cache: True # set cache responses to True, litellm defaults to using a redis cache
+ cache: True # set cache responses to True, litellm defaults to using a redis cache
cache_params:
type: qdrant-semantic
qdrant_semantic_cache_embedding_model: openai-embedding # the model should be defined on the model_list
qdrant_collection_name: test_collection
qdrant_quantization_config: binary
- similarity_threshold: 0.8 # similarity threshold for semantic cache
+ similarity_threshold: 0.8 # similarity threshold for semantic cache
```
#### Step 2: Add Qdrant Credentials to your .env
@@ -332,11 +336,11 @@ QDRANT_API_BASE = "https://5392d382-45*********.cloud.qdrant.io"
```
#### Step 3: Run proxy with config
+
```shell
$ litellm --config /path/to/config.yaml
```
-
#### Step 4. Test it
```shell
@@ -351,13 +355,15 @@ curl -i http://localhost:4000/v1/chat/completions \
}'
```
-**Expect to see `x-litellm-semantic-similarity` in the response headers when semantic caching is one**
+**Expect to see `x-litellm-semantic-similarity` in the response headers when semantic caching is
+one**
#### Step 1: Add `cache` to the config.yaml
+
```yaml
model_list:
- model_name: gpt-3.5-turbo
@@ -369,28 +375,70 @@ model_list:
litellm_settings:
set_verbose: True
- cache: True # set cache responses to True
- cache_params: # set cache params for s3
+ cache: True # set cache responses to True
+ cache_params: # set cache params for s3
type: s3
- s3_bucket_name: cache-bucket-litellm # AWS Bucket Name for S3
- s3_region_name: us-west-2 # AWS Region Name for S3
- s3_aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID # us os.environ/ to pass environment variables. This is AWS Access Key ID for S3
- s3_aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY # AWS Secret Access Key for S3
- s3_endpoint_url: https://s3.amazonaws.com # [OPTIONAL] S3 endpoint URL, if you want to use Backblaze/cloudflare s3 buckets
+ s3_bucket_name: cache-bucket-litellm # AWS Bucket Name for S3
+ s3_region_name: us-west-2 # AWS Region Name for S3
+ s3_aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID # us os.environ/ to pass environment variables. This is AWS Access Key ID for S3
+ s3_aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY # AWS Secret Access Key for S3
+ s3_endpoint_url: https://s3.amazonaws.com # [OPTIONAL] S3 endpoint URL, if you want to use Backblaze/cloudflare s3 buckets
```
#### Step 2: Run proxy with config
+
```shell
$ litellm --config /path/to/config.yaml
```
+
+
+
+#### Step 1: Add `cache` to the config.yaml
+
+```yaml
+model_list:
+ - model_name: gpt-3.5-turbo
+ litellm_params:
+ model: gpt-3.5-turbo
+ - model_name: text-embedding-ada-002
+ litellm_params:
+ model: text-embedding-ada-002
+
+litellm_settings:
+ set_verbose: True
+ cache: True # set cache responses to True
+ cache_params: # set cache params for gcs
+ type: gcs
+ gcs_bucket_name: cache-bucket-litellm # GCS Bucket Name for caching
+ gcs_path_service_account: os.environ/GCS_PATH_SERVICE_ACCOUNT # use os.environ/ to pass environment variables. This is the path to your GCS service account JSON file
+ gcs_path: cache/ # [OPTIONAL] GCS path prefix for cache objects
+```
+
+#### Step 2: Add GCS Credentials to .env
+
+Set the GCS environment variables in your .env file:
+
+```shell
+GCS_BUCKET_NAME="your-gcs-bucket-name"
+GCS_PATH_SERVICE_ACCOUNT="/path/to/service-account.json"
+```
+
+#### Step 3: Run proxy with config
+
+```shell
+$ litellm --config /path/to/config.yaml
+```
+
+
Caching can be enabled by adding the `cache` key in the `config.yaml`
#### Step 1: Add `cache` to the config.yaml
+
```yaml
model_list:
- model_name: gpt-3.5-turbo
@@ -405,40 +453,45 @@ model_list:
litellm_settings:
set_verbose: True
- cache: True # set cache responses to True
+ cache: True # set cache responses to True
cache_params:
- type: "redis-semantic"
- similarity_threshold: 0.8 # similarity threshold for semantic cache
+ type: "redis-semantic"
+ similarity_threshold: 0.8 # similarity threshold for semantic cache
redis_semantic_cache_embedding_model: azure-embedding-model # set this to a model_name set in model_list
```
#### Step 2: Add Redis Credentials to .env
+
Set either `REDIS_URL` or the `REDIS_HOST` in your os environment, to enable caching.
- ```shell
- REDIS_URL = "" # REDIS_URL='redis://username:password@hostname:port/database'
- ## OR ##
- REDIS_HOST = "" # REDIS_HOST='redis-18841.c274.us-east-1-3.ec2.cloud.redislabs.com'
- REDIS_PORT = "" # REDIS_PORT='18841'
- REDIS_PASSWORD = "" # REDIS_PASSWORD='liteLlmIsAmazing'
- ```
+```shell
+REDIS_URL = "" # REDIS_URL='redis://username:password@hostname:port/database'
+## OR ##
+REDIS_HOST = "" # REDIS_HOST='redis-18841.c274.us-east-1-3.ec2.cloud.redislabs.com'
+REDIS_PORT = "" # REDIS_PORT='18841'
+REDIS_PASSWORD = "" # REDIS_PASSWORD='liteLlmIsAmazing'
+```
**Additional kwargs**
-You can pass in any additional redis.Redis arg, by storing the variable + value in your os environment, like this:
+You can pass in any additional redis.Redis arg, by storing the variable + value in your os
+environment, like this:
+
```shell
REDIS_ = ""
-```
+```
#### Step 3: Run proxy with config
+
```shell
$ litellm --config /path/to/config.yaml
```
-
+
#### Step 1: Add `cache` to the config.yaml
+
```yaml
litellm_settings:
cache: True
@@ -447,6 +500,7 @@ litellm_settings:
```
#### Step 2: Run proxy with config
+
```shell
$ litellm --config /path/to/config.yaml
```
@@ -456,15 +510,17 @@ $ litellm --config /path/to/config.yaml
#### Step 1: Add `cache` to the config.yaml
+
```yaml
litellm_settings:
cache: True
cache_params:
type: disk
- disk_cache_dir: /tmp/litellm-cache # OPTIONAL, default to ./.litellm_cache
+ disk_cache_dir: /tmp/litellm-cache # OPTIONAL, default to ./.litellm_cache
```
#### Step 2: Run proxy with config
+
```shell
$ litellm --config /path/to/config.yaml
```
@@ -473,7 +529,6 @@ $ litellm --config /path/to/config.yaml
-
## Usage
### Basic
@@ -482,6 +537,7 @@ $ litellm --config /path/to/config.yaml
Send the same request twice:
+
```shell
curl http://0.0.0.0:4000/v1/chat/completions \
-H "Content-Type: application/json" \
@@ -499,10 +555,12 @@ curl http://0.0.0.0:4000/v1/chat/completions \
"temperature": 0.7
}'
```
+
Send the same request twice:
+
```shell
curl --location 'http://0.0.0.0:4000/embeddings' \
--header 'Content-Type: application/json' \
@@ -518,18 +576,19 @@ curl --location 'http://0.0.0.0:4000/embeddings' \
"input": ["write a litellm poem"]
}'
```
+
### Dynamic Cache Controls
-| Parameter | Type | Description |
-|-----------|------|-------------|
-| `ttl` | *Optional(int)* | Will cache the response for the user-defined amount of time (in seconds) |
-| `s-maxage` | *Optional(int)* | Will only accept cached responses that are within user-defined range (in seconds) |
-| `no-cache` | *Optional(bool)* | Will not store the response in cache. |
-| `no-store` | *Optional(bool)* | Will not cache the response |
-| `namespace` | *Optional(str)* | Will cache the response under a user-defined namespace |
+| Parameter | Type | Description |
+| ----------- | ---------------- | --------------------------------------------------------------------------------- |
+| `ttl` | _Optional(int)_ | Will cache the response for the user-defined amount of time (in seconds) |
+| `s-maxage` | _Optional(int)_ | Will only accept cached responses that are within user-defined range (in seconds) |
+| `no-cache` | _Optional(bool)_ | Will not store the response in cache. |
+| `no-store` | _Optional(bool)_ | Will not cache the response |
+| `namespace` | _Optional(str)_ | Will cache the response under a user-defined namespace |
Each cache parameter can be controlled on a per-request basis. Here are examples for each parameter:
@@ -558,6 +617,7 @@ chat_completion = client.chat.completions.create(
}
)
```
+
@@ -574,6 +634,7 @@ curl http://localhost:4000/v1/chat/completions \
]
}'
```
+
@@ -602,6 +663,7 @@ chat_completion = client.chat.completions.create(
}
)
```
+
@@ -618,10 +680,12 @@ curl http://localhost:4000/v1/chat/completions \
]
}'
```
+
### `no-cache`
+
Force a fresh response, bypassing the cache.
@@ -645,6 +709,7 @@ chat_completion = client.chat.completions.create(
}
)
```
+
@@ -661,6 +726,7 @@ curl http://localhost:4000/v1/chat/completions \
]
}'
```
+
@@ -668,7 +734,6 @@ curl http://localhost:4000/v1/chat/completions \
Will not store the response in cache.
-
@@ -690,6 +755,7 @@ chat_completion = client.chat.completions.create(
}
)
```
+
@@ -706,10 +772,12 @@ curl http://localhost:4000/v1/chat/completions \
]
}'
```
+
### `namespace`
+
Store the response under a specific cache namespace.
@@ -733,6 +801,7 @@ chat_completion = client.chat.completions.create(
}
)
```
+
@@ -749,36 +818,37 @@ curl http://localhost:4000/v1/chat/completions \
]
}'
```
+
-
-
## Set cache for proxy, but not on the actual llm api call
-Use this if you just want to enable features like rate limiting, and loadbalancing across multiple instances.
-
-Set `supported_call_types: []` to disable caching on the actual api call.
+Use this if you just want to enable features like rate limiting, and loadbalancing across multiple
+instances.
+Set `supported_call_types: []` to disable caching on the actual api call.
```yaml
litellm_settings:
cache: True
cache_params:
type: redis
- supported_call_types: []
+ supported_call_types: []
```
-
## Debugging Caching - `/cache/ping`
+
LiteLLM Proxy exposes a `/cache/ping` endpoint to test if the cache is working as expected
**Usage**
+
```shell
curl --location 'http://0.0.0.0:4000/cache/ping' -H "Authorization: Bearer sk-1234"
```
**Expected Response - when cache healthy**
+
```shell
{
"status": "healthy",
@@ -803,7 +873,8 @@ curl --location 'http://0.0.0.0:4000/cache/ping' -H "Authorization: Bearer sk-1
### Control Call Types Caching is on for - (`/chat/completion`, `/embeddings`, etc.)
-By default, caching is on for all call types. You can control which call types caching is on for by setting `supported_call_types` in `cache_params`
+By default, caching is on for all call types. You can control which call types caching is on for by
+setting `supported_call_types` in `cache_params`
**Cache will only be on for the call types specified in `supported_call_types`**
@@ -812,10 +883,13 @@ litellm_settings:
cache: True
cache_params:
type: redis
- supported_call_types: ["acompletion", "atext_completion", "aembedding", "atranscription"]
- # /chat/completions, /completions, /embeddings, /audio/transcriptions
+ supported_call_types:
+ ["acompletion", "atext_completion", "aembedding", "atranscription"]
+ # /chat/completions, /completions, /embeddings, /audio/transcriptions
```
+
### Set Cache Params on config.yaml
+
```yaml
model_list:
- model_name: gpt-3.5-turbo
@@ -827,22 +901,25 @@ model_list:
litellm_settings:
set_verbose: True
- cache: True # set cache responses to True, litellm defaults to using a redis cache
- cache_params: # cache_params are optional
- type: "redis" # The type of cache to initialize. Can be "local" or "redis". Defaults to "local".
- host: "localhost" # The host address for the Redis cache. Required if type is "redis".
- port: 6379 # The port number for the Redis cache. Required if type is "redis".
- password: "your_password" # The password for the Redis cache. Required if type is "redis".
-
+ cache: True # set cache responses to True, litellm defaults to using a redis cache
+ cache_params: # cache_params are optional
+ type: "redis" # The type of cache to initialize. Can be "local", "redis", "s3", or "gcs". Defaults to "local".
+ host: "localhost" # The host address for the Redis cache. Required if type is "redis".
+ port: 6379 # The port number for the Redis cache. Required if type is "redis".
+ password: "your_password" # The password for the Redis cache. Required if type is "redis".
+
# Optional configurations
- supported_call_types: ["acompletion", "atext_completion", "aembedding", "atranscription"]
- # /chat/completions, /completions, /embeddings, /audio/transcriptions
+ supported_call_types:
+ ["acompletion", "atext_completion", "aembedding", "atranscription"]
+ # /chat/completions, /completions, /embeddings, /audio/transcriptions
```
-### Deleting Cache Keys - `/cache/delete`
+### Deleting Cache Keys - `/cache/delete`
+
In order to delete a cache key, send a request to `/cache/delete` with the `keys` you want to delete
-Example
+Example
+
```shell
curl -X POST "http://0.0.0.0:4000/cache/delete" \
-H "Authorization: Bearer sk-1234" \
@@ -854,7 +931,10 @@ curl -X POST "http://0.0.0.0:4000/cache/delete" \
```
#### Viewing Cache Keys from responses
-You can view the cache_key in the response headers, on cache hits the cache key is sent as the `x-litellm-cache-key` response headers
+
+You can view the cache_key in the response headers, on cache hits the cache key is sent as the
+`x-litellm-cache-key` response headers
+
```shell
curl -i --location 'http://0.0.0.0:4000/chat/completions' \
--header 'Authorization: Bearer sk-1234' \
@@ -871,7 +951,8 @@ curl -i --location 'http://0.0.0.0:4000/chat/completions' \
}'
```
-Response from litellm proxy
+Response from litellm proxy
+
```json
date: Thu, 04 Apr 2024 17:37:21 GMT
content-type: application/json
@@ -891,7 +972,7 @@ x-litellm-cache-key: 586bf3f3c1bf5aecb55bd9996494d3bbc69eb58397163add6d49537762a
],
"created": 1712252235,
}
-
+
```
### **Set Caching Default Off - Opt in only **
@@ -916,7 +997,6 @@ litellm_settings:
2. **Opting in to cache when cache is default off**
-
@@ -939,6 +1019,7 @@ chat_completion = client.chat.completions.create(
}
)
```
+
@@ -977,45 +1058,49 @@ litellm_settings:
```yaml
cache_params:
- # ttl
+ # ttl
ttl: Optional[float]
default_in_memory_ttl: Optional[float]
default_in_redis_ttl: Optional[float]
max_connections: Optional[Int]
- # Type of cache (options: "local", "redis", "s3")
+ # Type of cache (options: "local", "redis", "s3", "gcs")
type: s3
# List of litellm call types to cache for
# Options: "completion", "acompletion", "embedding", "aembedding"
- supported_call_types: ["acompletion", "atext_completion", "aembedding", "atranscription"]
- # /chat/completions, /completions, /embeddings, /audio/transcriptions
+ supported_call_types:
+ ["acompletion", "atext_completion", "aembedding", "atranscription"]
+ # /chat/completions, /completions, /embeddings, /audio/transcriptions
# Redis cache parameters
- host: localhost # Redis server hostname or IP address
- port: "6379" # Redis server port (as a string)
- password: secret_password # Redis server password
+ host: localhost # Redis server hostname or IP address
+ port: "6379" # Redis server port (as a string)
+ password: secret_password # Redis server password
namespace: Optional[str] = None,
-
+
# GCP IAM Authentication for Redis
- gcp_service_account: "projects/-/serviceAccounts/your-sa@project.iam.gserviceaccount.com" # GCP service account for IAM authentication
- gcp_ssl_ca_certs: "./server-ca.pem" # Path to SSL CA certificate file for GCP Memorystore Redis
- ssl: true # Enable SSL for secure connections
- ssl_cert_reqs: null # Set to null for self-signed certificates
- ssl_check_hostname: false # Set to false for self-signed certificates
-
+ gcp_service_account: "projects/-/serviceAccounts/your-sa@project.iam.gserviceaccount.com" # GCP service account for IAM authentication
+ gcp_ssl_ca_certs: "./server-ca.pem" # Path to SSL CA certificate file for GCP Memorystore Redis
+ ssl: true # Enable SSL for secure connections
+ ssl_cert_reqs: null # Set to null for self-signed certificates
+ ssl_check_hostname: false # Set to false for self-signed certificates
# S3 cache parameters
- s3_bucket_name: your_s3_bucket_name # Name of the S3 bucket
- s3_region_name: us-west-2 # AWS region of the S3 bucket
- s3_api_version: 2006-03-01 # AWS S3 API version
- s3_use_ssl: true # Use SSL for S3 connections (options: true, false)
- s3_verify: true # SSL certificate verification for S3 connections (options: true, false)
- s3_endpoint_url: https://s3.amazonaws.com # S3 endpoint URL
- s3_aws_access_key_id: your_access_key # AWS Access Key ID for S3
- s3_aws_secret_access_key: your_secret_key # AWS Secret Access Key for S3
- s3_aws_session_token: your_session_token # AWS Session Token for temporary credentials
+ s3_bucket_name: your_s3_bucket_name # Name of the S3 bucket
+ s3_region_name: us-west-2 # AWS region of the S3 bucket
+ s3_api_version: 2006-03-01 # AWS S3 API version
+ s3_use_ssl: true # Use SSL for S3 connections (options: true, false)
+ s3_verify: true # SSL certificate verification for S3 connections (options: true, false)
+ s3_endpoint_url: https://s3.amazonaws.com # S3 endpoint URL
+ s3_aws_access_key_id: your_access_key # AWS Access Key ID for S3
+ s3_aws_secret_access_key: your_secret_key # AWS Secret Access Key for S3
+ s3_aws_session_token: your_session_token # AWS Session Token for temporary credentials
+ # GCS cache parameters
+ gcs_bucket_name: your_gcs_bucket_name # Name of the GCS bucket
+ gcs_path_service_account: /path/to/service-account.json # Path to GCS service account JSON file
+ gcs_path: cache/ # [OPTIONAL] GCS path prefix for cache objects
```
## Provider-Specific Optional Parameters Caching
diff --git a/docs/my-website/docs/proxy/config_settings.md b/docs/my-website/docs/proxy/config_settings.md
index f4359a86ba9..bd098e90573 100644
--- a/docs/my-website/docs/proxy/config_settings.md
+++ b/docs/my-website/docs/proxy/config_settings.md
@@ -24,9 +24,8 @@ litellm_settings:
turn_off_message_logging: boolean # prevent the messages and responses from being logged to on your callbacks, but request metadata will still be logged. Useful for privacy/compliance when handling sensitive data.
redact_user_api_key_info: boolean # Redact information about the user api key (hashed token, user_id, team id, etc.), from logs. Currently supported for Langfuse, OpenTelemetry, Logfire, ArizeAI logging.
langfuse_default_tags: ["cache_hit", "cache_key", "proxy_base_url", "user_api_key_alias", "user_api_key_user_id", "user_api_key_user_email", "user_api_key_team_alias", "semantic-similarity", "proxy_base_url"] # default tags for Langfuse Logging
-
# Networking settings
- request_timeout: 10 # (int) llm requesttimeout in seconds. Raise Timeout error if call takes longer than 10s. Sets litellm.request_timeout
+ request_timeout: 10 # (int) llm requesttimeout in seconds. Raise Timeout error if call takes longer than 10s. Sets litellm.request_timeout
force_ipv4: boolean # If true, litellm will force ipv4 for all LLM requests. Some users have seen httpx ConnectionError when using ipv6 + Anthropic API
# Debugging - see debugging docs for more options
@@ -35,63 +34,71 @@ litellm_settings:
# Fallbacks, reliability
default_fallbacks: ["claude-opus"] # set default_fallbacks, in case a specific model group is misconfigured / bad.
- content_policy_fallbacks: [{"gpt-3.5-turbo-small": ["claude-opus"]}] # fallbacks for ContentPolicyErrors
- context_window_fallbacks: [{"gpt-3.5-turbo-small": ["gpt-3.5-turbo-large", "claude-opus"]}] # fallbacks for ContextWindowExceededErrors
+ content_policy_fallbacks: [{ "gpt-3.5-turbo-small": ["claude-opus"] }] # fallbacks for ContentPolicyErrors
+ context_window_fallbacks: [{ "gpt-3.5-turbo-small": ["gpt-3.5-turbo-large", "claude-opus"] }] # fallbacks for ContextWindowExceededErrors
# MCP Aliases - Map aliases to MCP server names for easier tool access
- mcp_aliases: { "github": "github_mcp_server", "zapier": "zapier_mcp_server", "deepwiki": "deepwiki_mcp_server" } # Maps friendly aliases to MCP server names. Only the first alias for each server is used
+ mcp_aliases: {
+ "github": "github_mcp_server",
+ "zapier": "zapier_mcp_server",
+ "deepwiki": "deepwiki_mcp_server",
+ } # Maps friendly aliases to MCP server names. Only the first alias for each server is used
# Caching settings
- cache: true
- cache_params: # set cache params for redis
- type: redis # type of cache to initialize
+ cache: true
+ cache_params: # set cache params for redis
+ type: redis # type of cache to initialize (options: "local", "redis", "s3", "gcs")
# Optional - Redis Settings
- host: "localhost" # The host address for the Redis cache. Required if type is "redis".
- port: 6379 # The port number for the Redis cache. Required if type is "redis".
- password: "your_password" # The password for the Redis cache. Required if type is "redis".
+ host: "localhost" # The host address for the Redis cache. Required if type is "redis".
+ port: 6379 # The port number for the Redis cache. Required if type is "redis".
+ password: "your_password" # The password for the Redis cache. Required if type is "redis".
namespace: "litellm.caching.caching" # namespace for redis cache
max_connections: 100 # [OPTIONAL] Set Maximum number of Redis connections. Passed directly to redis-py.
-
# Optional - Redis Cluster Settings
- redis_startup_nodes: [{"host": "127.0.0.1", "port": "7001"}]
+ redis_startup_nodes: [{ "host": "127.0.0.1", "port": "7001" }]
# Optional - Redis Sentinel Settings
service_name: "mymaster"
sentinel_nodes: [["localhost", 26379]]
# Optional - GCP IAM Authentication for Redis
- gcp_service_account: "projects/-/serviceAccounts/your-sa@project.iam.gserviceaccount.com" # GCP service account for IAM authentication
- gcp_ssl_ca_certs: "./server-ca.pem" # Path to SSL CA certificate file for GCP Memorystore Redis
- ssl: true # Enable SSL for secure connections
- ssl_cert_reqs: null # Set to null for self-signed certificates
- ssl_check_hostname: false # Set to false for self-signed certificates
+ gcp_service_account: "projects/-/serviceAccounts/your-sa@project.iam.gserviceaccount.com" # GCP service account for IAM authentication
+ gcp_ssl_ca_certs: "./server-ca.pem" # Path to SSL CA certificate file for GCP Memorystore Redis
+ ssl: true # Enable SSL for secure connections
+ ssl_cert_reqs: null # Set to null for self-signed certificates
+ ssl_check_hostname: false # Set to false for self-signed certificates
# Optional - Qdrant Semantic Cache Settings
qdrant_semantic_cache_embedding_model: openai-embedding # the model should be defined on the model_list
qdrant_collection_name: test_collection
qdrant_quantization_config: binary
- similarity_threshold: 0.8 # similarity threshold for semantic cache
+ similarity_threshold: 0.8 # similarity threshold for semantic cache
# Optional - S3 Cache Settings
- s3_bucket_name: cache-bucket-litellm # AWS Bucket Name for S3
- s3_region_name: us-west-2 # AWS Region Name for S3
- s3_aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID # us os.environ/ to pass environment variables. This is AWS Access Key ID for S3
- s3_aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY # AWS Secret Access Key for S3
- s3_endpoint_url: https://s3.amazonaws.com # [OPTIONAL] S3 endpoint URL, if you want to use Backblaze/cloudflare s3 bucket
+ s3_bucket_name: cache-bucket-litellm # AWS Bucket Name for S3
+ s3_region_name: us-west-2 # AWS Region Name for S3
+ s3_aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID # us os.environ/ to pass environment variables. This is AWS Access Key ID for S3
+ s3_aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY # AWS Secret Access Key for S3
+ s3_endpoint_url: https://s3.amazonaws.com # [OPTIONAL] S3 endpoint URL, if you want to use Backblaze/cloudflare s3 bucket
+
+ # Optional - GCS Cache Settings
+ gcs_bucket_name: cache-bucket-litellm # GCS Bucket Name for caching
+ gcs_path_service_account: os.environ/GCS_PATH_SERVICE_ACCOUNT # Path to GCS service account JSON file
+ gcs_path: cache/ # [OPTIONAL] GCS path prefix for cache objects
# Common Cache settings
# Optional - Supported call types for caching
- supported_call_types: ["acompletion", "atext_completion", "aembedding", "atranscription"]
- # /chat/completions, /completions, /embeddings, /audio/transcriptions
+ supported_call_types:
+ ["acompletion", "atext_completion", "aembedding", "atranscription"]
+ # /chat/completions, /completions, /embeddings, /audio/transcriptions
mode: default_off # if default_off, you need to opt in to caching on a per call basis
ttl: 600 # ttl for caching
- disable_copilot_system_to_assistant: False # If false (default), converts all 'system' role messages to 'assistant' for GitHub Copilot compatibility. Set to true to disable this behavior.
-
+ disable_copilot_system_to_assistant: False # If false (default), converts all 'system' role messages to 'assistant' for GitHub Copilot compatibility. Set to true to disable this behavior.
callback_settings:
otel:
- message_logging: boolean # OTEL logging callback specific settings
+ message_logging: boolean # OTEL logging callback specific settings
general_settings:
completion_model: string
@@ -120,8 +127,8 @@ general_settings:
allow_requests_on_db_unavailable: boolean # if true, will allow requests that can not connect to the DB to verify Virtual Key to still work
custom_auth: string
- max_parallel_requests: 0 # the max parallel requests allowed per deployment
- global_max_parallel_requests: 0 # the max parallel requests allowed on the proxy all up
+ max_parallel_requests: 0 # the max parallel requests allowed per deployment
+ global_max_parallel_requests: 0 # the max parallel requests allowed on the proxy all up
infer_model_from_keys: true
background_health_checks: true
health_check_interval: 300
@@ -266,13 +273,14 @@ router_settings:
| forward_openai_org_id | boolean | If true, forwards the OpenAI Organization ID to the backend LLM call (if it's OpenAI). |
| forward_client_headers_to_llm_api | boolean | If true, forwards the client headers (any `x-` headers and `anthropic-beta` headers) to the backend LLM call |
| maximum_spend_logs_retention_period | str | Used to set the max retention time for spend logs in the db, after which they will be auto-purged |
-| maximum_spend_logs_retention_interval | str | Used to set the interval in which the spend log cleanup task should run in. |
+| maximum_spend_logs_retention_interval | str | Used to set the interval in which the spend log cleanup task should run in. |
+
### router_settings - Reference
:::info
-Most values can also be set via `litellm_settings`. If you see overlapping values, settings on `router_settings` will override those on `litellm_settings`.
-:::
+Most values can also be set via `litellm_settings`. If you see overlapping values, settings on
+`router_settings` will override those on `litellm_settings`. :::
```yaml
router_settings:
@@ -280,10 +288,10 @@ router_settings:
redis_host: # string
redis_password: # string
redis_port: # string
- enable_pre_call_checks: true # bool - Before call is made check if a call is within model context window
- allowed_fails: 3 # cooldown model if it fails > 1 call in a minute.
+ enable_pre_call_checks: true # bool - Before call is made check if a call is within model context window
+ allowed_fails: 3 # cooldown model if it fails > 1 call in a minute.
cooldown_time: 30 # (in seconds) how long to cooldown model if fails/min > allowed_fails
- disable_cooldowns: True # bool - Disable cooldowns for all models
+ disable_cooldowns: True # bool - Disable cooldowns for all models
enable_tag_filtering: True # bool - Use tag based routing for requests
retry_policy: { # Dict[str, int]: retry policy for different types of exceptions
"AuthenticationErrorRetries": 3,
@@ -294,11 +302,11 @@ router_settings:
}
allowed_fails_policy: {
"BadRequestErrorAllowedFails": 1000, # Allow 1000 BadRequestErrors before cooling down a deployment
- "AuthenticationErrorAllowedFails": 10, # int
- "TimeoutErrorAllowedFails": 12, # int
- "RateLimitErrorAllowedFails": 10000, # int
- "ContentPolicyViolationErrorAllowedFails": 15, # int
- "InternalServerErrorAllowedFails": 20, # int
+ "AuthenticationErrorAllowedFails": 10, # int
+ "TimeoutErrorAllowedFails": 12, # int
+ "RateLimitErrorAllowedFails": 10000, # int
+ "ContentPolicyViolationErrorAllowedFails": 15, # int
+ "InternalServerErrorAllowedFails": 20, # int
}
content_policy_fallbacks=[{"claude-2": ["my-fallback-model"]}] # List[Dict[str, List[str]]]: Fallback model for content policy violations
fallbacks=[{"claude-2": ["my-fallback-model"]}] # List[Dict[str, List[str]]]: Fallback model for all errors