mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
fix(caching): warn when provider-specific param is dropped from cache key
Provider-specific optional params (e.g. ollama num_ctx) are excluded from the cache key by default, so requests differing only in such a param collide and the second gets the first's cached response, silently. Emit a one-time warning per param name when this drop happens. Non-breaking; default behavior unchanged. Closes #42400
This commit is contained in:
parent
071cb49d32
commit
38800f1b70
2 changed files with 43 additions and 0 deletions
|
|
@ -38,6 +38,7 @@ from .redis_cluster_cache import RedisClusterCache
|
|||
from .redis_semantic_cache import RedisSemanticCache
|
||||
from .s3_cache import S3Cache
|
||||
|
||||
_warned_dropped_cache_params: set = set()
|
||||
|
||||
def print_verbose(print_statement):
|
||||
try:
|
||||
|
|
@ -365,6 +366,18 @@ class Cache:
|
|||
continue # ignore None params
|
||||
param_value = kwargs[param]
|
||||
cache_key += f"{param}: {param_value}"
|
||||
elif kwargs[param] is not None and param not in _warned_dropped_cache_params:
|
||||
# Provider-specific param (e.g. ollama num_ctx) is being excluded
|
||||
# from the cache key. Two requests differing only in this param
|
||||
# will collide and the second gets the first's cached response.
|
||||
# Warn once per param; set enable_caching_on_provider_specific_optional_params=True to include it.
|
||||
_warned_dropped_cache_params.add(param)
|
||||
verbose_logger.warning(
|
||||
"litellm.cache: provider-specific param '%s' is excluded from the cache key by default, "
|
||||
"so requests differing only in '%s' will return the same cached response. "
|
||||
"Set litellm.enable_caching_on_provider_specific_optional_params=True to include it in the key.",
|
||||
param, param,
|
||||
)
|
||||
|
||||
if is_semantic_cache:
|
||||
cache_key += self._get_semantic_cache_tenant_scope(kwargs)
|
||||
|
|
|
|||
|
|
@ -13,6 +13,7 @@ import hashlib
|
|||
import random
|
||||
|
||||
import pytest
|
||||
import logging
|
||||
|
||||
import litellm
|
||||
from litellm import aembedding, completion, embedding
|
||||
|
|
@ -63,6 +64,35 @@ async def test_dual_cache_async_batch_get_cache():
|
|||
await dual_cache.async_batch_get_cache(keys=["test_value", "test_value_2"])
|
||||
|
||||
assert mock_redis_cache.call_count == 1
|
||||
|
||||
|
||||
def test_cache_key_warns_on_dropped_provider_specific_param(caplog):
|
||||
"""Provider-specific params dropped from the cache key (default flag) should
|
||||
emit a one-time warning. Regression for #42400."""
|
||||
import litellm
|
||||
from litellm.caching.caching import Cache, _warned_dropped_cache_params
|
||||
|
||||
litellm.enable_caching_on_provider_specific_optional_params = False
|
||||
_warned_dropped_cache_params.discard("num_ctx") # reset in case another test warned
|
||||
cache = Cache()
|
||||
|
||||
with caplog.at_level(logging.WARNING):
|
||||
cache.get_cache_key(
|
||||
model="ollama/llama3.2",
|
||||
messages=[{"role": "user", "content": "hello"}],
|
||||
num_ctx=2048,
|
||||
)
|
||||
assert any("num_ctx" in r.message for r in caplog.records)
|
||||
|
||||
# second call must NOT warn again (one-time behavior)
|
||||
caplog.clear()
|
||||
with caplog.at_level(logging.WARNING):
|
||||
cache.get_cache_key(
|
||||
model="ollama/llama3.2",
|
||||
messages=[{"role": "user", "content": "hello"}],
|
||||
num_ctx=4096,
|
||||
)
|
||||
assert not any("num_ctx" in r.message for r in caplog.records)
|
||||
|
||||
|
||||
def test_dual_cache_batch_get_cache():
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue