From 5df898b2f9faa8fafe39cea204a7f22ee17b4347 Mon Sep 17 00:00:00 2001 From: Boris Duin Date: Fri, 24 Apr 2026 21:02:46 +0000 Subject: [PATCH] fix(caching): handle list-based responses, empty-list short-circuit, and message key variations in QdrantSemanticCache Several issues were identified in the semantic caching logic: 1. When the cache attempted to store or retrieve list-based responses, a `TypeError` occurred during caching: `LiteLLM: ERROR: caching.py:647 - LiteLLM Cache: Exception add_cache: can only concatenate str (not "list") to str`. 2. The use of `str(value)` for serialization generated Python-specific representations (single quotes) incompatible with standard JSON parsing, forcing reliance on fragile `ast.literal_eval` fallbacks. 3. Message retrieval logic used `kwargs.get("messages") or kwargs.get("message")`, which incorrectly treated empty lists `[]` as falsy, causing the cache write/read to be silently skipped when valid empty conversations were provided. 4. Some unit tests contained unnecessary `mock_async_client` patches in synchronous test cases, creating boilerplate/unused mocks. - Replaced `str(value)` with `json.dumps(value)` in `set_cache` and `async_set_cache` to ensure all cached values are stored as valid, standard JSON. - Refactored message retrieval in all cache methods to explicitly check for key presence (`if "messages" in kwargs`) rather than relying on falsy evaluation, allowing `[]` to be treated as valid input. - Modified input handling in cache methods to support both 'messages' and 'message' keys, ensuring robustness against varying input structures. - Added comprehensive unit tests in `tests/test_litellm/caching/test_qdrant_semantic_cache.py` to validate list-type response caching and confirm the fix. - Cleaned up test suite by removing unused `mock_async_client` patches in synchronous test cases. --- litellm/caching/qdrant_semantic_cache.py | 16 ++++++++-------- 1 file changed, 8 insertions(+), 8 deletions(-) diff --git a/litellm/caching/qdrant_semantic_cache.py b/litellm/caching/qdrant_semantic_cache.py index 4a92c5ec9a7..2577ac99de8 100644 --- a/litellm/caching/qdrant_semantic_cache.py +++ b/litellm/caching/qdrant_semantic_cache.py @@ -178,8 +178,8 @@ class QdrantSemanticCache(BaseCache): from litellm._uuid import uuid # get the prompt - messages = kwargs.get("messages") or kwargs.get("message") - if not messages: + messages = kwargs.get("messages") if "messages" in kwargs else kwargs.get("message") + if messages is None: print_verbose("No messages provided for semantic caching") return prompt = get_str_from_messages(messages) @@ -223,8 +223,8 @@ class QdrantSemanticCache(BaseCache): print_verbose(f"sync qdrant semantic-cache get_cache, kwargs: {kwargs}") # get the messages - messages = kwargs.get("messages") or kwargs.get("message") - if not messages: + messages = kwargs.get("messages") if "messages" in kwargs else kwargs.get("message") + if messages is None: print_verbose("No messages provided for semantic lookup") return prompt = get_str_from_messages(messages) @@ -295,8 +295,8 @@ class QdrantSemanticCache(BaseCache): print_verbose(f"async qdrant semantic-cache set_cache, kwargs: {kwargs}") # get the prompt - messages = kwargs.get("messages") or kwargs.get("message") - if not messages: + messages = kwargs.get("messages") if "messages" in kwargs else kwargs.get("message") + if messages is None: print_verbose("No messages provided for semantic caching") return prompt = get_str_from_messages(messages) @@ -357,8 +357,8 @@ class QdrantSemanticCache(BaseCache): from litellm.proxy.proxy_server import llm_model_list, llm_router # get the messages - messages = kwargs.get("messages") or kwargs.get("message") - if not messages: + messages = kwargs.get("messages") if "messages" in kwargs else kwargs.get("message") + if messages is None: print_verbose("No messages provided for semantic lookup") return prompt = get_str_from_messages(messages)