litellm/tests/test_litellm/caching/test_llm_caching_handler.py
yuneng-jiang 6a0d03914c
test: drop the cwd-relative sys.path.insert calls from the test suite (#37802)
* test: drop the cwd-relative sys.path.insert calls from the test suite

TQ003 stands at 1,077 across 1,058 files, and 1,015 of them are the same shape:
sys.path.insert(0, os.path.abspath("../..")) and its deeper siblings. The
argument resolves against the working directory rather than the file, so from
the repo root, where every job runs pytest, it inserts the directory two levels
above the checkout. It has never pointed at litellm. The package is installed
into the environment anyway, which is what actually makes the import work, and
what the rule's message has said all along.

Removing them leaves 1,634 imports of sys and os with no remaining reference,
and those go too, except where another test module imports the name back out of
the file. The rest of TQ003 is 62 call sites that resolve against __file__ or a
variable, which are a different question and are left alone.

Collection is identical either way: 45,871 tests and the same 51 pre-existing
collection errors before and after, and ruff reports no new undefined name.

* test: drop the duplicate imports the sys.path sweep exposed to F811

* test(pre-call-utils): restore the os import the new bedrock tests need
2026-08-22 09:25:58 -07:00

235 lines
6.8 KiB
Python

"""
Tests for LLMClientCache.
The cache intentionally does NOT close clients on eviction because evicted
clients may still be referenced by in-flight requests. Closing them eagerly
causes ``RuntimeError: Cannot send a request, as the client has been closed.``
See: https://github.com/BerriAI/litellm/pull/22247
"""
import asyncio
import warnings
import pytest
from litellm.caching.evicted_client_closer import EvictedClientCloser
from litellm.caching.llm_caching_handler import LLMClientCache
class MockAsyncClient:
"""Mock async HTTP client with an async close method."""
def __init__(self):
self.closed = False
async def close(self):
self.closed = True
class MockSyncClient:
"""Mock sync HTTP client with a sync close method."""
def __init__(self):
self.closed = False
def close(self):
self.closed = True
@pytest.mark.asyncio
async def test_remove_key_does_not_close_async_client():
"""
Evicting an async client from LLMClientCache must NOT close it because
an in-flight request may still hold a reference to the client.
Regression test for production 'client has been closed' crashes.
"""
cache = LLMClientCache(max_size_in_memory=2)
mock_client = MockAsyncClient()
cache.cache_dict["test-key"] = mock_client
cache.ttl_dict["test-key"] = 0 # expired
cache._remove_key("test-key")
# Give the event loop a chance to run any background tasks
await asyncio.sleep(0.1)
# Client must NOT be closed — it may still be in use
assert mock_client.closed is False
assert "test-key" not in cache.cache_dict
assert "test-key" not in cache.ttl_dict
def test_remove_key_does_not_close_sync_client():
"""
Evicting a sync client from the cache must NOT close it.
"""
cache = LLMClientCache(max_size_in_memory=2)
mock_client = MockSyncClient()
cache.cache_dict["test-key"] = mock_client
cache.ttl_dict["test-key"] = 0
cache._remove_key("test-key")
assert mock_client.closed is False
assert "test-key" not in cache.cache_dict
@pytest.mark.asyncio
async def test_eviction_does_not_close_async_clients():
"""
When the cache is full and an entry is evicted, the evicted async client
must remain open and must not produce 'coroutine was never awaited' warnings.
"""
cache = LLMClientCache(max_size_in_memory=2, default_ttl=1)
clients = []
for i in range(2):
client = MockAsyncClient()
clients.append(client)
cache.set_cache(f"key-{i}", client)
with warnings.catch_warnings(record=True) as caught_warnings:
warnings.simplefilter("always")
# This should trigger eviction of one of the existing entries
cache.set_cache("key-new", "new-value")
await asyncio.sleep(0.1)
coroutine_warnings = [
w for w in caught_warnings if "coroutine" in str(w.message).lower()
]
assert (
len(coroutine_warnings) == 0
), f"Got unawaited coroutine warnings: {coroutine_warnings}"
# Evicted clients must NOT be closed
for client in clients:
assert client.closed is False
@pytest.mark.asyncio
async def test_eviction_no_unawaited_coroutine_warning():
"""
Evicting an async client from LLMClientCache must not produce
'coroutine was never awaited' warnings.
Regression test for https://github.com/BerriAI/litellm/issues/22128
"""
cache = LLMClientCache(max_size_in_memory=2)
mock_client = MockAsyncClient()
cache.cache_dict["test-key"] = mock_client
cache.ttl_dict["test-key"] = 0 # expired
with warnings.catch_warnings(record=True) as caught_warnings:
warnings.simplefilter("always")
cache._remove_key("test-key")
await asyncio.sleep(0.1)
coroutine_warnings = [
w for w in caught_warnings if "coroutine" in str(w.message).lower()
]
assert (
len(coroutine_warnings) == 0
), f"Got unawaited coroutine warnings: {coroutine_warnings}"
def test_remove_key_no_event_loop():
"""
_remove_key works correctly even when there's no running event loop.
"""
cache = LLMClientCache(max_size_in_memory=2)
mock_client = MockAsyncClient()
cache.cache_dict["test-key"] = mock_client
cache.ttl_dict["test-key"] = 0
# Should not raise even though there's no running event loop
cache._remove_key("test-key")
assert "test-key" not in cache.cache_dict
class _FakeClock:
"""Hand-advanced monotonic clock, so grace windows need no real waiting."""
def __init__(self) -> None:
self.now = 1000.0
def __call__(self) -> float:
return self.now
def advance(self, seconds: float) -> None:
self.now += seconds
@pytest.mark.asyncio
async def test_evicted_litellm_owned_client_is_closed_once_the_grace_window_elapses():
"""
Eviction only drops the cache's reference. The SDK clients are reference
cycles, so without an explicit close the client keeps its connection pool
open until a generational collection runs.
"""
clock = _FakeClock()
cache = LLMClientCache(
max_size_in_memory=2,
evicted_client_closer=EvictedClientCloser(grace_seconds=60.0, clock=clock),
)
client = MockAsyncClient()
cache.set_cache("client-key", client, litellm_owned_client=True, ttl=600)
cache.ttl_dict = {key: 0 for key in cache.ttl_dict}
cache.expiration_heap = [(0, key) for _, key in cache.expiration_heap]
cache.evict_cache()
await asyncio.sleep(0.1)
assert client.closed is False, "an in-flight request may still hold the client"
clock.advance(61.0)
cache.get_cache("any-key")
await asyncio.sleep(0.1)
assert client.closed is True
@pytest.mark.asyncio
async def test_evicted_caller_supplied_client_is_never_closed():
"""litellm does not own a client the caller passed in, so it must stay open."""
clock = _FakeClock()
cache = LLMClientCache(
max_size_in_memory=2,
evicted_client_closer=EvictedClientCloser(grace_seconds=60.0, clock=clock),
)
client = MockAsyncClient()
cache.set_cache("client-key", client, ttl=600)
cache.ttl_dict = {key: 0 for key in cache.ttl_dict}
cache.expiration_heap = [(0, key) for _, key in cache.expiration_heap]
cache.evict_cache()
clock.advance(3600.0)
cache.get_cache("any-key")
await asyncio.sleep(0.1)
assert client.closed is False
def test_remove_key_removes_plain_values():
"""
_remove_key correctly removes non-client values (strings, dicts, etc.).
"""
cache = LLMClientCache(max_size_in_memory=5)
cache.cache_dict["str-key"] = "hello"
cache.ttl_dict["str-key"] = 0
cache.cache_dict["dict-key"] = {"foo": "bar"}
cache.ttl_dict["dict-key"] = 0
cache._remove_key("str-key")
cache._remove_key("dict-key")
assert "str-key" not in cache.cache_dict
assert "dict-key" not in cache.cache_dict