mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-01 02:02:20 +00:00
* refactor(cache): organize v2 cache as a package * docs: clarify experimental v2 guidance * fix(cache): verify cache-hit accounting and preserve logging metadata * refactor(cache): separate execution facts from host accounting * refactor(rust): build messages routes with named dependencies * wip * fix(cache): preserve facade policy and preflight fallback * refactor(cache): defer shared Python logging changes * test(gateway-inference): allow dead code in shared test helpers Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(cache): key prepared requests and honor facade controls * feat(cache): use Python caches from Rust Messages inference * refactor(cache): separate native and Python cache adapters * refactor(cache): enforce shared composition and adapter boundaries * fix(cache): let Python key delegated Rust Messages entries --------- Co-authored-by: Yujong Lee <yujong@berri.ai> Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
54 lines
1.7 KiB
Python
54 lines
1.7 KiB
Python
from __future__ import annotations
|
|
|
|
from dataclasses import dataclass
|
|
from typing import Final, Protocol, TypeAlias
|
|
from uuid import uuid4
|
|
|
|
from litellm.caching.caching import Cache
|
|
from litellm.rust_bridge import _native
|
|
from litellm.rust_bridge.response_cache import ResponseCacheRuntime
|
|
|
|
|
|
CacheRuntime: TypeAlias = _native._ResponseCacheRuntime # pyright: ignore[reportPrivateUsage] # private runtime under test
|
|
|
|
|
|
class CacheNamespace(Protocol):
|
|
@property
|
|
def cache(self) -> object: ...
|
|
|
|
|
|
@dataclass(frozen=True, slots=True)
|
|
class CacheTestResolver:
|
|
namespace: CacheNamespace
|
|
|
|
def resolve(self) -> CacheRuntime:
|
|
return CacheRuntime.from_selected(self.namespace.cache)
|
|
|
|
|
|
class CacheLookup(Protocol):
|
|
def get_cache(self, **kwargs: object) -> object: ...
|
|
def flush_cache(self) -> object: ...
|
|
|
|
|
|
def request(key: str = "key") -> dict[str, object]:
|
|
return {"key": {"preset": key}}
|
|
|
|
|
|
def native_runtime(facade: Cache) -> CacheRuntime:
|
|
return CacheRuntime.from_cache(facade)
|
|
|
|
|
|
def activate_native(facade: Cache) -> Cache:
|
|
facade._native_cache = ResponseCacheRuntime(_native._ResponseCacheRuntime.from_cache(facade)) # pyright: ignore[reportPrivateUsage] # explicitly select the runtime under test
|
|
return facade
|
|
|
|
|
|
def assert_native_runtime(facade: Cache) -> ResponseCacheRuntime:
|
|
runtime: Final = facade._native_cache # pyright: ignore[reportPrivateUsage] # the activation under test has no public accessor
|
|
assert isinstance(runtime, ResponseCacheRuntime)
|
|
assert runtime.kind == "native"
|
|
return runtime
|
|
|
|
|
|
def completion_kwargs(label: str) -> dict[str, object]:
|
|
return {"model": "gpt-4o", "messages": [{"role": "user", "content": f"{label} {uuid4().hex}"}]}
|