[Fix] x-litellm-cache-key header not being returned on cache hit (#15348)

* fix: x-cache-key

* test_cache_key_in_hidden_params_acompletion

* fix: remove_cache_control_flag_from_messages_and_tools
This commit is contained in:
Ishaan Jaff 2025-10-08 18:10:43 -07:00 • committed by GitHub
parent 97031dc8ee
commit 2f42c806cb
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
3 changed files with 62 additions and 15 deletions

View file

@ -14,10 +14,10 @@ It utilizes the (RedisCache, s3Cache, RedisSemanticCache, QdrantSemanticCache, I
In each method it will call the appropriate method from caching.py
"""
import time
import asyncio
import datetime
import inspect
import time
from typing import (
TYPE_CHECKING,
Any,
@ -62,12 +62,10 @@ else:
LiteLLMLoggingObj = Any
from litellm.litellm_core_utils.streaming_handler import CustomStreamWrapper
from litellm.litellm_core_utils.core_helpers import (
_get_parent_otel_span_from_kwargs,
_get_parent_otel_span_from_kwargs,
)
from litellm.litellm_core_utils.streaming_handler import CustomStreamWrapper
class CachingHandlerResponse(BaseModel):
@ -214,9 +212,7 @@ class LLMCachingHandler:
end_time=end_time,
cache_hit=cache_hit,
)
cache_key = litellm.cache._get_preset_cache_key_from_kwargs(
**kwargs
)
cache_key = litellm.cache.get_cache_key(**kwargs)
if (
isinstance(cached_result, BaseModel)
or isinstance(cached_result, CustomStreamWrapper)
@ -330,9 +326,7 @@ class LLMCachingHandler:
end_time=end_time,
cache_hit=cache_hit
)
cache_key = litellm.cache._get_preset_cache_key_from_kwargs(
**kwargs
)
cache_key = litellm.cache.get_cache_key(**kwargs)
if (
isinstance(cached_result, BaseModel)
or isinstance(cached_result, CustomStreamWrapper)

View file

@ -397,13 +397,13 @@ class OpenAIGPTConfig(BaseLLMModelInfo, BaseConfig):
)
from litellm.types.llms.openai import ChatCompletionToolParam
for message in messages:
message = cast(
for i, message in enumerate(messages):
messages[i] = cast(
AllMessageValues, filter_value_from_dict(message, "cache_control") # type: ignore
)
if tools is not None:
for tool in tools:
tool = cast(
for i, tool in enumerate(tools):
tools[i] = cast(
ChatCompletionToolParam,
filter_value_from_dict(tool, "cache_control"), # type: ignore
)

View file

@ -2765,3 +2765,56 @@ def test_caching_thinking_args_hit(): # test in memory cache
except Exception as e:
print(f"error occurred: {traceback.format_exc()}")
pytest.fail(f"Error occurred: {e}")
@pytest.mark.asyncio
async def test_cache_key_in_hidden_params_acompletion():
"""
Test that cache_key is present in _hidden_params on cache hits for acompletion.
Validates fix for missing x-litellm-cache-key header on proxy cache hits.
"""
litellm.cache = Cache(
type="redis",
host=os.environ["REDIS_HOST"],
port=os.environ["REDIS_PORT"],
password=os.environ["REDIS_PASSWORD"],
)
unique_content = f"test cache key hidden params {uuid.uuid4()}"
messages = [{"role": "user", "content": unique_content}]
# First call - cache miss
response1 = await litellm.acompletion(
model="gpt-3.5-turbo",
messages=messages,
mock_response="test response",
caching=True,
)
print(f"Response 1 _hidden_params: {response1._hidden_params}")
assert response1._hidden_params.get("cache_hit") is not True
await asyncio.sleep(0.5)
# Second call - cache hit
response2 = await litellm.acompletion(
model="gpt-3.5-turbo",
messages=messages,
mock_response="test response",
caching=True,
)
print(f"Response 2 _hidden_params: {response2._hidden_params}")
# Verify cache hit occurred
assert response2._hidden_params.get("cache_hit") is True
# Verify cache_key is present in _hidden_params
assert "cache_key" in response2._hidden_params
assert response2._hidden_params["cache_key"] is not None
# Verify both responses have same ID (cache hit)
assert response1.id == response2.id
litellm.cache = None