mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
fix(purview): make protection scope cache true LRU on hits
OrderedDict.get() does not update insertion order; call move_to_end on TTL-valid cache hits so popitem(last=False) evicts least-recently-used users instead of FIFO by first insert. Add a regression test with a small max cache size. Co-authored-by: Sameer Kankute <Sameerlite@users.noreply.github.com>
This commit is contained in:
parent
f09e335375
commit
ce10c1b304
2 changed files with 43 additions and 4 deletions
|
|
@ -1,7 +1,7 @@
|
|||
import time
|
||||
import uuid
|
||||
from collections import OrderedDict
|
||||
from typing import TYPE_CHECKING, Any, Dict, List, MutableMapping, Optional, Tuple
|
||||
from typing import TYPE_CHECKING, Any, Dict, List, Optional, Tuple
|
||||
|
||||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.litellm_core_utils.prompt_templates.common_utils import (
|
||||
|
|
@ -61,7 +61,9 @@ class PurviewGuardrailBase:
|
|||
|
||||
# Protection scope cache: user_id -> (etag, scope_response, fetched_at)
|
||||
# Capped at 1000 entries (LRU eviction) to avoid unbounded growth.
|
||||
self._scope_cache: MutableMapping[str, Tuple[str, Dict, float]] = OrderedDict()
|
||||
self._scope_cache: OrderedDict[str, Tuple[str, Dict[str, Any], float]] = (
|
||||
OrderedDict()
|
||||
)
|
||||
self._scope_cache_maxsize = 1000
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
|
|
@ -145,6 +147,7 @@ class PurviewGuardrailBase:
|
|||
now = time.time()
|
||||
|
||||
if cached and (now - cached[2]) < SCOPE_CACHE_TTL_SECONDS:
|
||||
self._scope_cache.move_to_end(user_id)
|
||||
return cached[0], cached[1]
|
||||
|
||||
url = (
|
||||
|
|
@ -165,9 +168,9 @@ class PurviewGuardrailBase:
|
|||
etag = response_headers.get("etag", response_headers.get("ETag", ""))
|
||||
|
||||
self._scope_cache[user_id] = (etag, response_json, now)
|
||||
# Evict oldest entry when cache exceeds max size.
|
||||
# Evict least-recently-used entry when cache exceeds max size.
|
||||
while len(self._scope_cache) > self._scope_cache_maxsize:
|
||||
self._scope_cache.popitem(last=False) # type: ignore[attr-defined]
|
||||
self._scope_cache.popitem(last=False)
|
||||
return etag, response_json
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
|
|
|
|||
|
|
@ -531,6 +531,42 @@ class TestScopeCaching:
|
|||
assert etag2 == "scope-etag"
|
||||
assert mock_post.call_count == 1
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_scope_cache_lru_keeps_hot_user_on_eviction(self):
|
||||
"""Frequently accessed users should not be evicted before cold entries."""
|
||||
guardrail = _make_guardrail()
|
||||
guardrail._scope_cache_maxsize = 3
|
||||
|
||||
scope_payload = (
|
||||
{
|
||||
"value": [
|
||||
{"activities": "uploadText", "executionMode": "evaluateInline"}
|
||||
]
|
||||
},
|
||||
{"ETag": "scope-etag"},
|
||||
)
|
||||
|
||||
with patch.object(
|
||||
guardrail, "_graph_post", new_callable=AsyncMock
|
||||
) as mock_post:
|
||||
mock_post.return_value = scope_payload
|
||||
|
||||
await guardrail._compute_protection_scopes("user-a")
|
||||
await guardrail._compute_protection_scopes("user-b")
|
||||
await guardrail._compute_protection_scopes("user-c")
|
||||
assert mock_post.call_count == 3
|
||||
|
||||
await guardrail._compute_protection_scopes("user-a")
|
||||
assert mock_post.call_count == 3
|
||||
|
||||
await guardrail._compute_protection_scopes("user-d")
|
||||
assert mock_post.call_count == 4
|
||||
|
||||
await guardrail._compute_protection_scopes("user-a")
|
||||
assert mock_post.call_count == 4
|
||||
assert "user-a" in guardrail._scope_cache
|
||||
assert "user-b" not in guardrail._scope_cache
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_scope_invalidated_on_modified(self):
|
||||
guardrail = _make_guardrail()
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue