fix(caching): warn when provider-specific param is dropped from cache key

Provider-specific optional params (e.g. ollama num_ctx) are excluded from
the cache key by default, so requests differing only in such a param collide
and the second gets the first's cached response, silently. Emit a one-time
warning per param name when this drop happens. Non-breaking; default
behavior unchanged.

Closes #42400
This commit is contained in:
Amr Mohammed 2026-09-27 16:04:11 +03:00
parent 071cb49d32
commit 38800f1b70
2 changed files with 43 additions and 0 deletions

View file

@ -38,6 +38,7 @@ from .redis_cluster_cache import RedisClusterCache
from .redis_semantic_cache import RedisSemanticCache
from .s3_cache import S3Cache
_warned_dropped_cache_params: set = set()
def print_verbose(print_statement):
try:
@ -365,6 +366,18 @@ class Cache:
continue # ignore None params
param_value = kwargs[param]
cache_key += f"{param}: {param_value}"
elif kwargs[param] is not None and param not in _warned_dropped_cache_params:
# Provider-specific param (e.g. ollama num_ctx) is being excluded
# from the cache key. Two requests differing only in this param
# will collide and the second gets the first's cached response.
# Warn once per param; set enable_caching_on_provider_specific_optional_params=True to include it.
_warned_dropped_cache_params.add(param)
verbose_logger.warning(
"litellm.cache: provider-specific param '%s' is excluded from the cache key by default, "
"so requests differing only in '%s' will return the same cached response. "
"Set litellm.enable_caching_on_provider_specific_optional_params=True to include it in the key.",
param, param,
)
if is_semantic_cache:
cache_key += self._get_semantic_cache_tenant_scope(kwargs)

View file

@ -13,6 +13,7 @@ import hashlib
import random
import pytest
import logging
import litellm
from litellm import aembedding, completion, embedding
@ -63,6 +64,35 @@ async def test_dual_cache_async_batch_get_cache():
await dual_cache.async_batch_get_cache(keys=["test_value", "test_value_2"])
assert mock_redis_cache.call_count == 1
def test_cache_key_warns_on_dropped_provider_specific_param(caplog):
"""Provider-specific params dropped from the cache key (default flag) should
emit a one-time warning. Regression for #42400."""
import litellm
from litellm.caching.caching import Cache, _warned_dropped_cache_params
litellm.enable_caching_on_provider_specific_optional_params = False
_warned_dropped_cache_params.discard("num_ctx") # reset in case another test warned
cache = Cache()
with caplog.at_level(logging.WARNING):
cache.get_cache_key(
model="ollama/llama3.2",
messages=[{"role": "user", "content": "hello"}],
num_ctx=2048,
)
assert any("num_ctx" in r.message for r in caplog.records)
# second call must NOT warn again (one-time behavior)
caplog.clear()
with caplog.at_level(logging.WARNING):
cache.get_cache_key(
model="ollama/llama3.2",
messages=[{"role": "user", "content": "hello"}],
num_ctx=4096,
)
assert not any("num_ctx" in r.message for r in caplog.records)
def test_dual_cache_batch_get_cache():