From 38800f1b709eb083be35874833bc7988de7b7606 Mon Sep 17 00:00:00 2001 From: Amr Mohammed Date: Sun, 27 Sep 2026 16:04:11 +0300 Subject: [PATCH] fix(caching): warn when provider-specific param is dropped from cache key Provider-specific optional params (e.g. ollama num_ctx) are excluded from the cache key by default, so requests differing only in such a param collide and the second gets the first's cached response, silently. Emit a one-time warning per param name when this drop happens. Non-breaking; default behavior unchanged. Closes #42400 --- litellm/caching/caching.py | 13 +++++++++++++ tests/local_testing/test_caching.py | 30 +++++++++++++++++++++++++++++ 2 files changed, 43 insertions(+) diff --git a/litellm/caching/caching.py b/litellm/caching/caching.py index 6a98b104221..70fd4ae3f66 100644 --- a/litellm/caching/caching.py +++ b/litellm/caching/caching.py @@ -38,6 +38,7 @@ from .redis_cluster_cache import RedisClusterCache from .redis_semantic_cache import RedisSemanticCache from .s3_cache import S3Cache +_warned_dropped_cache_params: set = set() def print_verbose(print_statement): try: @@ -365,6 +366,18 @@ class Cache: continue # ignore None params param_value = kwargs[param] cache_key += f"{param}: {param_value}" + elif kwargs[param] is not None and param not in _warned_dropped_cache_params: + # Provider-specific param (e.g. ollama num_ctx) is being excluded + # from the cache key. Two requests differing only in this param + # will collide and the second gets the first's cached response. + # Warn once per param; set enable_caching_on_provider_specific_optional_params=True to include it. + _warned_dropped_cache_params.add(param) + verbose_logger.warning( + "litellm.cache: provider-specific param '%s' is excluded from the cache key by default, " + "so requests differing only in '%s' will return the same cached response. " + "Set litellm.enable_caching_on_provider_specific_optional_params=True to include it in the key.", + param, param, + ) if is_semantic_cache: cache_key += self._get_semantic_cache_tenant_scope(kwargs) diff --git a/tests/local_testing/test_caching.py b/tests/local_testing/test_caching.py index f9deb9c100b..42323d23311 100644 --- a/tests/local_testing/test_caching.py +++ b/tests/local_testing/test_caching.py @@ -13,6 +13,7 @@ import hashlib import random import pytest +import logging import litellm from litellm import aembedding, completion, embedding @@ -63,6 +64,35 @@ async def test_dual_cache_async_batch_get_cache(): await dual_cache.async_batch_get_cache(keys=["test_value", "test_value_2"]) assert mock_redis_cache.call_count == 1 + + +def test_cache_key_warns_on_dropped_provider_specific_param(caplog): + """Provider-specific params dropped from the cache key (default flag) should + emit a one-time warning. Regression for #42400.""" + import litellm + from litellm.caching.caching import Cache, _warned_dropped_cache_params + + litellm.enable_caching_on_provider_specific_optional_params = False + _warned_dropped_cache_params.discard("num_ctx") # reset in case another test warned + cache = Cache() + + with caplog.at_level(logging.WARNING): + cache.get_cache_key( + model="ollama/llama3.2", + messages=[{"role": "user", "content": "hello"}], + num_ctx=2048, + ) + assert any("num_ctx" in r.message for r in caplog.records) + + # second call must NOT warn again (one-time behavior) + caplog.clear() + with caplog.at_level(logging.WARNING): + cache.get_cache_key( + model="ollama/llama3.2", + messages=[{"role": "user", "content": "hello"}], + num_ctx=4096, + ) + assert not any("num_ctx" in r.message for r in caplog.records) def test_dual_cache_batch_get_cache():