From 38800f1b709eb083be35874833bc7988de7b7606 Mon Sep 17 00:00:00 2001 From: Amr Mohammed Date: Sun, 27 Sep 2026 16:04:11 +0300 Subject: [PATCH 1/3] fix(caching): warn when provider-specific param is dropped from cache key Provider-specific optional params (e.g. ollama num_ctx) are excluded from the cache key by default, so requests differing only in such a param collide and the second gets the first's cached response, silently. Emit a one-time warning per param name when this drop happens. Non-breaking; default behavior unchanged. Closes #42400 --- litellm/caching/caching.py | 13 +++++++++++++ tests/local_testing/test_caching.py | 30 +++++++++++++++++++++++++++++ 2 files changed, 43 insertions(+) diff --git a/litellm/caching/caching.py b/litellm/caching/caching.py index 6a98b104221..70fd4ae3f66 100644 --- a/litellm/caching/caching.py +++ b/litellm/caching/caching.py @@ -38,6 +38,7 @@ from .redis_cluster_cache import RedisClusterCache from .redis_semantic_cache import RedisSemanticCache from .s3_cache import S3Cache +_warned_dropped_cache_params: set = set() def print_verbose(print_statement): try: @@ -365,6 +366,18 @@ class Cache: continue # ignore None params param_value = kwargs[param] cache_key += f"{param}: {param_value}" + elif kwargs[param] is not None and param not in _warned_dropped_cache_params: + # Provider-specific param (e.g. ollama num_ctx) is being excluded + # from the cache key. Two requests differing only in this param + # will collide and the second gets the first's cached response. + # Warn once per param; set enable_caching_on_provider_specific_optional_params=True to include it. + _warned_dropped_cache_params.add(param) + verbose_logger.warning( + "litellm.cache: provider-specific param '%s' is excluded from the cache key by default, " + "so requests differing only in '%s' will return the same cached response. " + "Set litellm.enable_caching_on_provider_specific_optional_params=True to include it in the key.", + param, param, + ) if is_semantic_cache: cache_key += self._get_semantic_cache_tenant_scope(kwargs) diff --git a/tests/local_testing/test_caching.py b/tests/local_testing/test_caching.py index f9deb9c100b..42323d23311 100644 --- a/tests/local_testing/test_caching.py +++ b/tests/local_testing/test_caching.py @@ -13,6 +13,7 @@ import hashlib import random import pytest +import logging import litellm from litellm import aembedding, completion, embedding @@ -63,6 +64,35 @@ async def test_dual_cache_async_batch_get_cache(): await dual_cache.async_batch_get_cache(keys=["test_value", "test_value_2"]) assert mock_redis_cache.call_count == 1 + + +def test_cache_key_warns_on_dropped_provider_specific_param(caplog): + """Provider-specific params dropped from the cache key (default flag) should + emit a one-time warning. Regression for #42400.""" + import litellm + from litellm.caching.caching import Cache, _warned_dropped_cache_params + + litellm.enable_caching_on_provider_specific_optional_params = False + _warned_dropped_cache_params.discard("num_ctx") # reset in case another test warned + cache = Cache() + + with caplog.at_level(logging.WARNING): + cache.get_cache_key( + model="ollama/llama3.2", + messages=[{"role": "user", "content": "hello"}], + num_ctx=2048, + ) + assert any("num_ctx" in r.message for r in caplog.records) + + # second call must NOT warn again (one-time behavior) + caplog.clear() + with caplog.at_level(logging.WARNING): + cache.get_cache_key( + model="ollama/llama3.2", + messages=[{"role": "user", "content": "hello"}], + num_ctx=4096, + ) + assert not any("num_ctx" in r.message for r in caplog.records) def test_dual_cache_batch_get_cache(): From c1f79fdf862a531761162be1e04a64893b4e5384 Mon Sep 17 00:00:00 2001 From: Amr Mohammed Date: Sun, 27 Sep 2026 16:30:31 +0300 Subject: [PATCH 2/3] refactor(caching): remove redundant comments, restore warning-set state in test --- litellm/caching/caching.py | 4 ---- tests/local_testing/test_caching.py | 33 ++++++++++++----------------- 2 files changed, 13 insertions(+), 24 deletions(-) diff --git a/litellm/caching/caching.py b/litellm/caching/caching.py index 50b29830f31..088922046e1 100644 --- a/litellm/caching/caching.py +++ b/litellm/caching/caching.py @@ -394,10 +394,6 @@ class Cache: param_value = kwargs[param] cache_key += f"{param}: {param_value}" elif kwargs[param] is not None and param not in _warned_dropped_cache_params: - # Provider-specific param (e.g. ollama num_ctx) is being excluded - # from the cache key. Two requests differing only in this param - # will collide and the second gets the first's cached response. - # Warn once per param; set enable_caching_on_provider_specific_optional_params=True to include it. _warned_dropped_cache_params.add(param) verbose_logger.warning( "litellm.cache: provider-specific param '%s' is excluded from the cache key by default, " diff --git a/tests/local_testing/test_caching.py b/tests/local_testing/test_caching.py index c6eda589096..8f809df457a 100644 --- a/tests/local_testing/test_caching.py +++ b/tests/local_testing/test_caching.py @@ -78,32 +78,25 @@ async def test_dual_cache_async_batch_get_cache(): def test_cache_key_warns_on_dropped_provider_specific_param(caplog): - """Provider-specific params dropped from the cache key (default flag) should - emit a one-time warning. Regression for #42400.""" import litellm from litellm.caching.caching import Cache, _warned_dropped_cache_params litellm.enable_caching_on_provider_specific_optional_params = False - _warned_dropped_cache_params.discard("num_ctx") # reset in case another test warned + _warned_dropped_cache_params.discard("num_ctx") cache = Cache() + try: + with caplog.at_level(logging.WARNING): + cache.get_cache_key(model="ollama/llama3.2", + messages=[{"role": "user", "content": "hello"}], num_ctx=2048) + assert any("num_ctx" in r.message for r in caplog.records) - with caplog.at_level(logging.WARNING): - cache.get_cache_key( - model="ollama/llama3.2", - messages=[{"role": "user", "content": "hello"}], - num_ctx=2048, - ) - assert any("num_ctx" in r.message for r in caplog.records) - - # second call must NOT warn again (one-time behavior) - caplog.clear() - with caplog.at_level(logging.WARNING): - cache.get_cache_key( - model="ollama/llama3.2", - messages=[{"role": "user", "content": "hello"}], - num_ctx=4096, - ) - assert not any("num_ctx" in r.message for r in caplog.records) + caplog.clear() + with caplog.at_level(logging.WARNING): + cache.get_cache_key(model="ollama/llama3.2", + messages=[{"role": "user", "content": "hello"}], num_ctx=4096) + assert not any("num_ctx" in r.message for r in caplog.records) + finally: + _warned_dropped_cache_params.discard("num_ctx") def test_dual_cache_batch_get_cache(): From 3cf31e0c4de52b48ed49fe33673576d8c28847c0 Mon Sep 17 00:00:00 2001 From: Amr Mohammed Date: Mon, 28 Sep 2026 10:02:16 +0300 Subject: [PATCH 3/3] fix(caching): bound warning registry to prevent unbounded growth --- litellm/caching/caching.py | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/litellm/caching/caching.py b/litellm/caching/caching.py index 088922046e1..4a7ed6a76e9 100644 --- a/litellm/caching/caching.py +++ b/litellm/caching/caching.py @@ -393,13 +393,18 @@ class Cache: continue # ignore None params param_value = kwargs[param] cache_key += f"{param}: {param_value}" - elif kwargs[param] is not None and param not in _warned_dropped_cache_params: + elif ( + kwargs[param] is not None + and param not in _warned_dropped_cache_params + and len(_warned_dropped_cache_params) < 100 + ): _warned_dropped_cache_params.add(param) verbose_logger.warning( "litellm.cache: provider-specific param '%s' is excluded from the cache key by default, " "so requests differing only in '%s' will return the same cached response. " "Set litellm.enable_caching_on_provider_specific_optional_params=True to include it in the key.", - param, param, + param, + param, ) if is_semantic_cache: