From 226953e1d8eb7f3b459ccb16e9e357e47a3420a0 Mon Sep 17 00:00:00 2001 From: Krrish Dholakia Date: Fri, 15 Mar 2024 14:40:11 -0700 Subject: [PATCH] feat(batch_redis_get.py): batch redis GET requests for a given key + call type reduces the number of GET requests we're making in high-throughput scenarios --- litellm/caching.py | 46 ++++++++- litellm/proxy/_new_secret_config.yaml | 8 +- litellm/proxy/hooks/batch_redis_get.py | 124 +++++++++++++++++++++++++ litellm/proxy/proxy_server.py | 10 ++ litellm/utils.py | 6 +- 5 files changed, 189 insertions(+), 5 deletions(-) create mode 100644 litellm/proxy/hooks/batch_redis_get.py diff --git a/litellm/caching.py b/litellm/caching.py index 9df95f199ed..eda4439412b 100644 --- a/litellm/caching.py +++ b/litellm/caching.py @@ -129,6 +129,16 @@ class RedisCache(BaseCache): f"LiteLLM Caching: set() - Got exception from REDIS : {str(e)}" ) + async def async_scan_iter(self, pattern: str, count: int = 100) -> list: + keys = [] + _redis_client = self.init_async_client() + async with _redis_client as redis_client: + async for key in redis_client.scan_iter(match=pattern + "*", count=count): + keys.append(key) + if len(keys) >= count: + break + return keys + async def async_set_cache(self, key, value, **kwargs): _redis_client = self.init_async_client() async with _redis_client as redis_client: @@ -172,8 +182,6 @@ class RedisCache(BaseCache): return results except Exception as e: print_verbose(f"Error occurred in pipeline write - {str(e)}") - # NON blocking - notify users Redis is throwing an exception - logging.debug("LiteLLM Caching: set() - Got exception from REDIS : ", e) def _get_cache_logic(self, cached_response: Any): """ @@ -220,6 +228,36 @@ class RedisCache(BaseCache): traceback.print_exc() logging.debug("LiteLLM Caching: get() - Got exception from REDIS: ", e) + async def async_get_cache_pipeline(self, key_list) -> dict: + """ + Use Redis for bulk read operations + """ + _redis_client = await self.init_async_client() + key_value_dict = {} + try: + async with _redis_client as redis_client: + async with redis_client.pipeline(transaction=True) as pipe: + # Queue the get operations in the pipeline for all keys. + for cache_key in key_list: + pipe.get(cache_key) # Queue GET command in pipeline + + # Execute the pipeline and await the results. + results = await pipe.execute() + + # Associate the results back with their keys. + # 'results' is a list of values corresponding to the order of keys in 'key_list'. + key_value_dict = dict(zip(key_list, results)) + + decoded_results = { + k.decode("utf-8"): self._get_cache_logic(v) + for k, v in key_value_dict.items() + } + + return decoded_results + except Exception as e: + print_verbose(f"Error occurred in pipeline read - {str(e)}") + return key_value_dict + def flush_cache(self): self.redis_client.flushall() @@ -1001,6 +1039,10 @@ class Cache: if self.namespace is not None: hash_hex = f"{self.namespace}:{hash_hex}" print_verbose(f"Hashed Key with Namespace: {hash_hex}") + elif kwargs.get("metadata", {}).get("redis_namespace", None) is not None: + _namespace = kwargs.get("metadata", {}).get("redis_namespace", None) + hash_hex = f"{_namespace}:{hash_hex}" + print_verbose(f"Hashed Key with Namespace: {hash_hex}") return hash_hex def generate_streaming_content(self, content): diff --git a/litellm/proxy/_new_secret_config.yaml b/litellm/proxy/_new_secret_config.yaml index aab9b3d5ca4..cb53b4baaf7 100644 --- a/litellm/proxy/_new_secret_config.yaml +++ b/litellm/proxy/_new_secret_config.yaml @@ -9,6 +9,12 @@ model_list: model: gpt-3.5-turbo-1106 api_key: os.environ/OPENAI_API_KEY +litellm_settings: + cache: true + cache_params: + type: redis + # callbacks: ["batch_redis_requests"] + general_settings: master_key: sk-1234 - database_url: "postgresql://krrishdholakia:9yQkKWiB8vVs@ep-icy-union-a5j4dwls.us-east-2.aws.neon.tech/neondb?sslmode=require" \ No newline at end of file + # database_url: "postgresql://krrishdholakia:9yQkKWiB8vVs@ep-icy-union-a5j4dwls.us-east-2.aws.neon.tech/neondb?sslmode=require" \ No newline at end of file diff --git a/litellm/proxy/hooks/batch_redis_get.py b/litellm/proxy/hooks/batch_redis_get.py new file mode 100644 index 00000000000..25589e0df88 --- /dev/null +++ b/litellm/proxy/hooks/batch_redis_get.py @@ -0,0 +1,124 @@ +# What this does? +## Gets a key's redis cache, and store it in memory for 1 minute. +## This reduces the number of REDIS GET requests made during high-traffic by the proxy. +### [BETA] this is in Beta. And might change. + +from typing import Optional, Literal +import litellm +from litellm.caching import DualCache, RedisCache, InMemoryCache +from litellm.proxy._types import UserAPIKeyAuth +from litellm.integrations.custom_logger import CustomLogger +from litellm._logging import verbose_proxy_logger +from fastapi import HTTPException +import json, traceback + + +class _PROXY_BatchRedisRequests(CustomLogger): + # Class variables or attributes + in_memory_cache: Optional[InMemoryCache] = None + + def __init__(self): + litellm.cache.async_get_cache = ( + self.async_get_cache + ) # map the litellm 'get_cache' function to our custom function + + def print_verbose( + self, print_statement, debug_level: Literal["INFO", "DEBUG"] = "DEBUG" + ): + if debug_level == "DEBUG": + verbose_proxy_logger.debug(print_statement) + elif debug_level == "INFO": + verbose_proxy_logger.debug(print_statement) + if litellm.set_verbose is True: + print(print_statement) # noqa + + async def async_pre_call_hook( + self, + user_api_key_dict: UserAPIKeyAuth, + cache: DualCache, + data: dict, + call_type: str, + ): + try: + """ + Get the user key + + Check if a key starting with `litellm:: 0: + key_value_dict = ( + await litellm.cache.cache.async_get_cache_pipeline( + key_list=keys + ) + ) + + ## Add to cache + for key, value in key_value_dict.items(): + _cache_key = f"{cache_key_name}:{key}" + cache.in_memory_cache.cache_dict[_cache_key] = value + + ## Set cache namespace if it's a miss + data["metadata"]["redis_namespace"] = cache_key_name + except HTTPException as e: + raise e + except Exception as e: + traceback.print_exc() + + async def async_get_cache(self, *args, **kwargs): + """ + - Check if the cache key is in-memory + + - Else return None + """ + try: # never block execution + if "cache_key" in kwargs: + cache_key = kwargs["cache_key"] + else: + cache_key = litellm.cache.get_cache_key( + *args, **kwargs + ) # returns ":" - we pass redis_namespace in async_pre_call_hook. Done to avoid rewriting the async_set_cache logic + if cache_key is not None and self.in_memory_cache is not None: + cache_control_args = kwargs.get("cache", {}) + max_age = cache_control_args.get( + "s-max-age", cache_control_args.get("s-maxage", float("inf")) + ) + cached_result = self.in_memory_cache.get_cache( + cache_key, *args, **kwargs + ) + return litellm.cache._get_cache_logic( + cached_result=cached_result, max_age=max_age + ) + except Exception as e: + return None diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index e3e778dc7c5..2326caa27cf 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -1795,6 +1795,16 @@ class ProxyConfig: _ENTERPRISE_PromptInjectionDetection() ) imported_list.append(prompt_injection_detection_obj) + elif ( + isinstance(callback, str) + and callback == "batch_redis_requests" + ): + from litellm.proxy.hooks.batch_redis_get import ( + _PROXY_BatchRedisRequests, + ) + + batch_redis_obj = _PROXY_BatchRedisRequests() + imported_list.append(batch_redis_obj) else: imported_list.append( get_instance_fn( diff --git a/litellm/utils.py b/litellm/utils.py index 7ad4107a98b..97a2bc14944 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -72,7 +72,7 @@ from .integrations.litedebugger import LiteDebugger from .proxy._types import KeyManagementSystem from openai import OpenAIError as OriginalError from openai._models import BaseModel as OpenAIObject -from .caching import S3Cache, RedisSemanticCache +from .caching import S3Cache, RedisSemanticCache, RedisCache from .exceptions import ( AuthenticationError, BadRequestError, @@ -2806,7 +2806,9 @@ def client(original_function): ): if len(cached_result) == 1 and cached_result[0] is None: cached_result = None - elif isinstance(litellm.cache.cache, RedisSemanticCache): + elif isinstance( + litellm.cache.cache, RedisSemanticCache + ) or isinstance(litellm.cache.cache, RedisCache): preset_cache_key = litellm.cache.get_cache_key(*args, **kwargs) kwargs["preset_cache_key"] = ( preset_cache_key # for streaming calls, we need to pass the preset_cache_key