mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-07 08:26:10 +00:00
* test: drop the cwd-relative sys.path.insert calls from the test suite
TQ003 stands at 1,077 across 1,058 files, and 1,015 of them are the same shape:
sys.path.insert(0, os.path.abspath("../..")) and its deeper siblings. The
argument resolves against the working directory rather than the file, so from
the repo root, where every job runs pytest, it inserts the directory two levels
above the checkout. It has never pointed at litellm. The package is installed
into the environment anyway, which is what actually makes the import work, and
what the rule's message has said all along.
Removing them leaves 1,634 imports of sys and os with no remaining reference,
and those go too, except where another test module imports the name back out of
the file. The rest of TQ003 is 62 call sites that resolve against __file__ or a
variable, which are a different question and are left alone.
Collection is identical either way: 45,871 tests and the same 51 pre-existing
collection errors before and after, and ruff reports no new undefined name.
* test: drop the duplicate imports the sys.path sweep exposed to F811
* test(pre-call-utils): restore the os import the new bedrock tests need
235 lines
6.8 KiB
Python
235 lines
6.8 KiB
Python
"""
|
|
Tests for LLMClientCache.
|
|
|
|
The cache intentionally does NOT close clients on eviction because evicted
|
|
clients may still be referenced by in-flight requests. Closing them eagerly
|
|
causes ``RuntimeError: Cannot send a request, as the client has been closed.``
|
|
|
|
See: https://github.com/BerriAI/litellm/pull/22247
|
|
"""
|
|
|
|
import asyncio
|
|
import warnings
|
|
|
|
import pytest
|
|
|
|
|
|
from litellm.caching.evicted_client_closer import EvictedClientCloser
|
|
from litellm.caching.llm_caching_handler import LLMClientCache
|
|
|
|
|
|
class MockAsyncClient:
|
|
"""Mock async HTTP client with an async close method."""
|
|
|
|
def __init__(self):
|
|
self.closed = False
|
|
|
|
async def close(self):
|
|
self.closed = True
|
|
|
|
|
|
class MockSyncClient:
|
|
"""Mock sync HTTP client with a sync close method."""
|
|
|
|
def __init__(self):
|
|
self.closed = False
|
|
|
|
def close(self):
|
|
self.closed = True
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_remove_key_does_not_close_async_client():
|
|
"""
|
|
Evicting an async client from LLMClientCache must NOT close it because
|
|
an in-flight request may still hold a reference to the client.
|
|
|
|
Regression test for production 'client has been closed' crashes.
|
|
"""
|
|
cache = LLMClientCache(max_size_in_memory=2)
|
|
|
|
mock_client = MockAsyncClient()
|
|
cache.cache_dict["test-key"] = mock_client
|
|
cache.ttl_dict["test-key"] = 0 # expired
|
|
|
|
cache._remove_key("test-key")
|
|
# Give the event loop a chance to run any background tasks
|
|
await asyncio.sleep(0.1)
|
|
|
|
# Client must NOT be closed — it may still be in use
|
|
assert mock_client.closed is False
|
|
assert "test-key" not in cache.cache_dict
|
|
assert "test-key" not in cache.ttl_dict
|
|
|
|
|
|
def test_remove_key_does_not_close_sync_client():
|
|
"""
|
|
Evicting a sync client from the cache must NOT close it.
|
|
"""
|
|
cache = LLMClientCache(max_size_in_memory=2)
|
|
|
|
mock_client = MockSyncClient()
|
|
cache.cache_dict["test-key"] = mock_client
|
|
cache.ttl_dict["test-key"] = 0
|
|
|
|
cache._remove_key("test-key")
|
|
|
|
assert mock_client.closed is False
|
|
assert "test-key" not in cache.cache_dict
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_eviction_does_not_close_async_clients():
|
|
"""
|
|
When the cache is full and an entry is evicted, the evicted async client
|
|
must remain open and must not produce 'coroutine was never awaited' warnings.
|
|
"""
|
|
cache = LLMClientCache(max_size_in_memory=2, default_ttl=1)
|
|
|
|
clients = []
|
|
for i in range(2):
|
|
client = MockAsyncClient()
|
|
clients.append(client)
|
|
cache.set_cache(f"key-{i}", client)
|
|
|
|
with warnings.catch_warnings(record=True) as caught_warnings:
|
|
warnings.simplefilter("always")
|
|
# This should trigger eviction of one of the existing entries
|
|
cache.set_cache("key-new", "new-value")
|
|
await asyncio.sleep(0.1)
|
|
|
|
coroutine_warnings = [
|
|
w for w in caught_warnings if "coroutine" in str(w.message).lower()
|
|
]
|
|
assert (
|
|
len(coroutine_warnings) == 0
|
|
), f"Got unawaited coroutine warnings: {coroutine_warnings}"
|
|
|
|
# Evicted clients must NOT be closed
|
|
for client in clients:
|
|
assert client.closed is False
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_eviction_no_unawaited_coroutine_warning():
|
|
"""
|
|
Evicting an async client from LLMClientCache must not produce
|
|
'coroutine was never awaited' warnings.
|
|
|
|
Regression test for https://github.com/BerriAI/litellm/issues/22128
|
|
"""
|
|
cache = LLMClientCache(max_size_in_memory=2)
|
|
|
|
mock_client = MockAsyncClient()
|
|
cache.cache_dict["test-key"] = mock_client
|
|
cache.ttl_dict["test-key"] = 0 # expired
|
|
|
|
with warnings.catch_warnings(record=True) as caught_warnings:
|
|
warnings.simplefilter("always")
|
|
cache._remove_key("test-key")
|
|
await asyncio.sleep(0.1)
|
|
|
|
coroutine_warnings = [
|
|
w for w in caught_warnings if "coroutine" in str(w.message).lower()
|
|
]
|
|
assert (
|
|
len(coroutine_warnings) == 0
|
|
), f"Got unawaited coroutine warnings: {coroutine_warnings}"
|
|
|
|
|
|
def test_remove_key_no_event_loop():
|
|
"""
|
|
_remove_key works correctly even when there's no running event loop.
|
|
"""
|
|
cache = LLMClientCache(max_size_in_memory=2)
|
|
|
|
mock_client = MockAsyncClient()
|
|
cache.cache_dict["test-key"] = mock_client
|
|
cache.ttl_dict["test-key"] = 0
|
|
|
|
# Should not raise even though there's no running event loop
|
|
cache._remove_key("test-key")
|
|
assert "test-key" not in cache.cache_dict
|
|
|
|
|
|
class _FakeClock:
|
|
"""Hand-advanced monotonic clock, so grace windows need no real waiting."""
|
|
|
|
def __init__(self) -> None:
|
|
self.now = 1000.0
|
|
|
|
def __call__(self) -> float:
|
|
return self.now
|
|
|
|
def advance(self, seconds: float) -> None:
|
|
self.now += seconds
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_evicted_litellm_owned_client_is_closed_once_the_grace_window_elapses():
|
|
"""
|
|
Eviction only drops the cache's reference. The SDK clients are reference
|
|
cycles, so without an explicit close the client keeps its connection pool
|
|
open until a generational collection runs.
|
|
"""
|
|
clock = _FakeClock()
|
|
cache = LLMClientCache(
|
|
max_size_in_memory=2,
|
|
evicted_client_closer=EvictedClientCloser(grace_seconds=60.0, clock=clock),
|
|
)
|
|
|
|
client = MockAsyncClient()
|
|
cache.set_cache("client-key", client, litellm_owned_client=True, ttl=600)
|
|
|
|
cache.ttl_dict = {key: 0 for key in cache.ttl_dict}
|
|
cache.expiration_heap = [(0, key) for _, key in cache.expiration_heap]
|
|
cache.evict_cache()
|
|
await asyncio.sleep(0.1)
|
|
assert client.closed is False, "an in-flight request may still hold the client"
|
|
|
|
clock.advance(61.0)
|
|
cache.get_cache("any-key")
|
|
await asyncio.sleep(0.1)
|
|
|
|
assert client.closed is True
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_evicted_caller_supplied_client_is_never_closed():
|
|
"""litellm does not own a client the caller passed in, so it must stay open."""
|
|
clock = _FakeClock()
|
|
cache = LLMClientCache(
|
|
max_size_in_memory=2,
|
|
evicted_client_closer=EvictedClientCloser(grace_seconds=60.0, clock=clock),
|
|
)
|
|
|
|
client = MockAsyncClient()
|
|
cache.set_cache("client-key", client, ttl=600)
|
|
|
|
cache.ttl_dict = {key: 0 for key in cache.ttl_dict}
|
|
cache.expiration_heap = [(0, key) for _, key in cache.expiration_heap]
|
|
cache.evict_cache()
|
|
|
|
clock.advance(3600.0)
|
|
cache.get_cache("any-key")
|
|
await asyncio.sleep(0.1)
|
|
|
|
assert client.closed is False
|
|
|
|
|
|
def test_remove_key_removes_plain_values():
|
|
"""
|
|
_remove_key correctly removes non-client values (strings, dicts, etc.).
|
|
"""
|
|
cache = LLMClientCache(max_size_in_memory=5)
|
|
|
|
cache.cache_dict["str-key"] = "hello"
|
|
cache.ttl_dict["str-key"] = 0
|
|
cache.cache_dict["dict-key"] = {"foo": "bar"}
|
|
cache.ttl_dict["dict-key"] = 0
|
|
|
|
cache._remove_key("str-key")
|
|
cache._remove_key("dict-key")
|
|
|
|
assert "str-key" not in cache.cache_dict
|
|
assert "dict-key" not in cache.cache_dict
|