mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-28 01:32:17 +00:00
* refactor(rust): align the cache crates with Python and activate every backend The cache port had drifted: lifecycle and Redis-only operations sat on `BaseCache`, counters were pinned to `f64`, each semantic backend defined its own embedder and prompt handling, and only the in-memory backend could be selected natively. - Split `disconnect` and `test_connection` out of `BaseCache` into optional capabilities, implemented only where the Python class defines them, and give every Redis-only operation its own capability trait. - Decouple counters from the stored value type, so one backend can serve both responses and counters as Python's `RedisCache` does. - Share one `Embedder` and prompt contract in `litellm_cache::semantic`, and make the Redis and Valkey semantic backends generic over their codec. - Port the Python operations that were missing: `async_refresh_ttl`, `async_rpush_and_trim`, `async_set_cache_pipeline_with_ttls`, the DualCache pipeline, sadd, bulk delete and TTL reads, and the semantic-similarity write-back. - Take the HTTP client from the host pool in the GCS, S3 and Azure backends. - Activate all nine backends through the Rust catalog, whose rules all stay `PYTHON_ONLY`, and route the `Cache` facade's storage calls to the native runtime when one is selected. - Give every crate the same layout, move all tests to `tests/` on rstest, and add the shared `litellm-cache-testing` contract suite. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> * fix: freeze native cache request kwargs and batch entries for type discipline Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix: declare semantic lookup methods in the native stub Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(rust): align the cache crates with Python and activate every backend The cache port had drifted: lifecycle and Redis-only operations sat on `BaseCache`, counters were pinned to `f64`, each semantic backend defined its own embedder and prompt handling, and only the in-memory backend could be selected natively. - Split `disconnect` and `test_connection` out of `BaseCache` into optional capabilities, implemented only where the Python class defines them, and give every Redis-only operation its own capability trait. - Decouple counters from the stored value type, so one backend can serve both responses and counters as Python's `RedisCache` does. - Share one `Embedder` and prompt contract in `litellm_cache::semantic`, and make the Redis and Valkey semantic backends generic over their codec. - Port the Python operations that were missing: `async_refresh_ttl`, `async_rpush_and_trim`, `async_set_cache_pipeline_with_ttls`, the DualCache pipeline, sadd, bulk delete and TTL reads, and the semantic-similarity write-back. - Take the HTTP client from the host pool in the GCS, S3 and Azure backends. - Activate all nine backends through the Rust catalog, whose rules all stay `PYTHON_ONLY`, and route the `Cache` facade's storage calls to the native runtime when one is selected. - Give every crate the same layout, move all tests to `tests/` on rstest, and add the shared `litellm-cache-testing` contract suite. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> * fix: freeze native cache request kwargs and batch entries for type discipline Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix: declare semantic lookup methods in the native stub Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(rust): opt the native Messages and tokenizer suites into Rust explicitly #42517 made the Messages, token counter and tokenizer routes Python-only, so tests/test_litellm_rust silently exercised the Python path or failed outright. Each suite now prepends a RUST_OPT_IN rule for its route, keeping native coverage without changing the shipped default. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> * fix(rust): pop one at a time in the Redis 6 lpop pipeline and drop explanatory comments Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: Yujong Lee <yujong@berri.ai> Co-authored-by: Claude Opus 5 <noreply@anthropic.com> Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
327 lines
12 KiB
Rust
327 lines
12 KiB
Rust
use litellm_cache_response::PartialHits;
|
|
use litellm_host_python::{ExecutionStep, from_py, release_gil, run_async, to_py};
|
|
use pyo3::{
|
|
PyTraverseError, PyVisit,
|
|
exceptions::{PyRuntimeError, PyValueError},
|
|
prelude::*,
|
|
types::PyDict,
|
|
};
|
|
use serde_json::Value;
|
|
|
|
use super::{
|
|
activation::activate,
|
|
cache_error,
|
|
callback::PythonCallback,
|
|
config::{CacheConfigProjection, NativeCacheConfig},
|
|
future::{ready_none, ready_value},
|
|
native::{NativeResponseCache, SemanticReply},
|
|
request::{now, request, requests},
|
|
};
|
|
use crate::errors::RustBridgeDeclined;
|
|
|
|
pub(super) enum CacheBinding {
|
|
Disabled,
|
|
Native(NativeResponseCache),
|
|
PythonCallback(PythonCallback),
|
|
}
|
|
|
|
#[pyclass(frozen, name = "_ResponseCacheRuntime")]
|
|
pub(crate) struct ResolvedCache {
|
|
binding: CacheBinding,
|
|
pid: u32,
|
|
}
|
|
|
|
impl ResolvedCache {
|
|
pub(super) fn new(binding: CacheBinding) -> Self {
|
|
Self {
|
|
binding,
|
|
pid: std::process::id(),
|
|
}
|
|
}
|
|
|
|
fn check_process(&self) -> PyResult<()> {
|
|
if matches!(self.binding, CacheBinding::Native(_)) && self.pid != std::process::id() {
|
|
return Err(PyRuntimeError::new_err(
|
|
"native cache bindings must be resolved again after fork",
|
|
));
|
|
}
|
|
Ok(())
|
|
}
|
|
|
|
pub(crate) fn lookup_step(
|
|
&self,
|
|
py: Python<'_>,
|
|
input: &Bound<'_, PyAny>,
|
|
kwargs: Option<&Bound<'_, PyDict>>,
|
|
) -> PyResult<ExecutionStep> {
|
|
self.check_process()?;
|
|
let awaitable = match &self.binding {
|
|
CacheBinding::Disabled => ready_none(py)?,
|
|
CacheBinding::Native(service) => {
|
|
let request = request(input)?;
|
|
service.async_lookup_py(py, request)?
|
|
}
|
|
CacheBinding::PythonCallback(callback) => callback.async_lookup(py, kwargs)?,
|
|
};
|
|
Ok(ExecutionStep::Await(awaitable.unbind()))
|
|
}
|
|
}
|
|
|
|
#[pymethods]
|
|
impl ResolvedCache {
|
|
#[staticmethod]
|
|
fn from_cache(cache: &Bound<'_, PyAny>) -> PyResult<Self> {
|
|
let config = match NativeCacheConfig::project(cache)? {
|
|
CacheConfigProjection::Native(config) => *config,
|
|
CacheConfigProjection::Unsupported(reason) => {
|
|
return Err(RustBridgeDeclined::new_err(reason.message()));
|
|
}
|
|
};
|
|
let backend = cache.getattr("cache")?;
|
|
let service = activate(cache.py(), &backend, config)?;
|
|
Ok(Self::new(CacheBinding::Native(service)))
|
|
}
|
|
|
|
#[getter]
|
|
fn kind(&self) -> &'static str {
|
|
match self.binding {
|
|
CacheBinding::Disabled => "disabled",
|
|
CacheBinding::Native(_) => "native",
|
|
CacheBinding::PythonCallback(_) => "python_callback",
|
|
}
|
|
}
|
|
|
|
#[pyo3(signature = (request, *, callback_kwargs=None))]
|
|
fn lookup(
|
|
&self,
|
|
py: Python<'_>,
|
|
request: &Bound<'_, PyAny>,
|
|
callback_kwargs: Option<&Bound<'_, PyDict>>,
|
|
) -> PyResult<Py<PyAny>> {
|
|
self.check_process()?;
|
|
match &self.binding {
|
|
CacheBinding::Disabled => Ok(py.None()),
|
|
CacheBinding::Native(service) => {
|
|
let request = self::request(request)?;
|
|
let service = service.clone();
|
|
let response = release_gil(py, move || service.lookup(&request, now()))
|
|
.map_err(cache_error)?;
|
|
to_py(py, &response)
|
|
}
|
|
CacheBinding::PythonCallback(callback) => {
|
|
callback.lookup(py, callback_kwargs).map(Bound::unbind)
|
|
}
|
|
}
|
|
}
|
|
|
|
/// `(response, similarity)`: the similarity is `None` when the backend reports none.
|
|
fn lookup_semantic(&self, py: Python<'_>, request: &Bound<'_, PyAny>) -> PyResult<Py<PyAny>> {
|
|
self.check_process()?;
|
|
match &self.binding {
|
|
CacheBinding::Native(service) => {
|
|
let request = self::request(request)?;
|
|
let service = service.clone();
|
|
let lookup = release_gil(py, move || service.lookup_semantic(&request, now()))
|
|
.map_err(cache_error)?;
|
|
to_py(py, &SemanticReply::from(lookup))
|
|
}
|
|
CacheBinding::Disabled => to_py(py, &SemanticReply(None, None)),
|
|
CacheBinding::PythonCallback(_) => Err(PyRuntimeError::new_err(
|
|
"semantic lookups require a native cache binding",
|
|
)),
|
|
}
|
|
}
|
|
|
|
fn async_lookup_semantic<'py>(
|
|
&self,
|
|
py: Python<'py>,
|
|
request: &Bound<'py, PyAny>,
|
|
) -> PyResult<Bound<'py, PyAny>> {
|
|
self.check_process()?;
|
|
match &self.binding {
|
|
CacheBinding::Native(service) => {
|
|
service.async_lookup_semantic_py(py, self::request(request)?)
|
|
}
|
|
CacheBinding::Disabled => ready_value(py, &SemanticReply(None, None)),
|
|
CacheBinding::PythonCallback(_) => Err(PyRuntimeError::new_err(
|
|
"semantic lookups require a native cache binding",
|
|
)),
|
|
}
|
|
}
|
|
|
|
#[pyo3(signature = (request, response, *, callback_kwargs=None))]
|
|
fn store(
|
|
&self,
|
|
py: Python<'_>,
|
|
request: &Bound<'_, PyAny>,
|
|
response: &Bound<'_, PyAny>,
|
|
callback_kwargs: Option<&Bound<'_, PyDict>>,
|
|
) -> PyResult<()> {
|
|
self.check_process()?;
|
|
match &self.binding {
|
|
CacheBinding::Disabled => Ok(()),
|
|
CacheBinding::Native(service) => {
|
|
let request = self::request(request)?;
|
|
let response: Value = from_py(response)?;
|
|
let service = service.clone();
|
|
release_gil(py, move || service.store(&request, response, now()))
|
|
.map_err(cache_error)
|
|
}
|
|
CacheBinding::PythonCallback(callback) => callback.store(py, response, callback_kwargs),
|
|
}
|
|
}
|
|
|
|
#[pyo3(signature = (requests, *, callback_kwargs=None))]
|
|
fn lookup_batch(
|
|
&self,
|
|
py: Python<'_>,
|
|
requests: &Bound<'_, PyAny>,
|
|
callback_kwargs: Option<&Bound<'_, PyAny>>,
|
|
) -> PyResult<Py<PyAny>> {
|
|
self.check_process()?;
|
|
match &self.binding {
|
|
CacheBinding::Disabled => {
|
|
let requests = self::requests(requests)?;
|
|
to_py(py, &PartialHits::new(vec![None; requests.len()]))
|
|
}
|
|
CacheBinding::Native(service) => {
|
|
let requests = self::requests(requests)?;
|
|
let service = service.clone();
|
|
let response = release_gil(py, move || service.lookup_batch(&requests, now()))
|
|
.map_err(cache_error)?;
|
|
to_py(py, &response)
|
|
}
|
|
CacheBinding::PythonCallback(callback) => callback
|
|
.lookup_batch(py, requests, callback_kwargs)
|
|
.map(Bound::unbind),
|
|
}
|
|
}
|
|
|
|
#[pyo3(signature = (request, *, callback_kwargs=None))]
|
|
fn async_lookup<'py>(
|
|
&self,
|
|
py: Python<'py>,
|
|
request: &Bound<'py, PyAny>,
|
|
callback_kwargs: Option<&Bound<'py, PyDict>>,
|
|
) -> PyResult<Bound<'py, PyAny>> {
|
|
let ExecutionStep::Await(awaitable) = self.lookup_step(py, request, callback_kwargs)?
|
|
else {
|
|
unreachable!()
|
|
};
|
|
Ok(awaitable.into_bound(py))
|
|
}
|
|
|
|
#[pyo3(signature = (request, response, *, callback_kwargs=None))]
|
|
fn async_store<'py>(
|
|
&self,
|
|
py: Python<'py>,
|
|
request: &Bound<'py, PyAny>,
|
|
response: &Bound<'py, PyAny>,
|
|
callback_kwargs: Option<&Bound<'py, PyDict>>,
|
|
) -> PyResult<Bound<'py, PyAny>> {
|
|
self.check_process()?;
|
|
match &self.binding {
|
|
CacheBinding::Disabled => ready_none(py),
|
|
CacheBinding::Native(service) => {
|
|
let request = self::request(request)?;
|
|
let response: Value = from_py(response)?;
|
|
service.async_store_py(py, request, response)
|
|
}
|
|
CacheBinding::PythonCallback(callback) => {
|
|
callback.async_store(py, response, callback_kwargs)
|
|
}
|
|
}
|
|
}
|
|
|
|
#[pyo3(signature = (requests, *, callback_kwargs=None))]
|
|
fn async_lookup_batch<'py>(
|
|
&self,
|
|
py: Python<'py>,
|
|
requests: &Bound<'py, PyAny>,
|
|
callback_kwargs: Option<&Bound<'py, PyAny>>,
|
|
) -> PyResult<Bound<'py, PyAny>> {
|
|
self.check_process()?;
|
|
match &self.binding {
|
|
CacheBinding::Disabled => {
|
|
let requests = self::requests(requests)?;
|
|
ready_value(py, &PartialHits::new(vec![None; requests.len()]))
|
|
}
|
|
CacheBinding::Native(service) => {
|
|
let requests = self::requests(requests)?;
|
|
let service = service.clone();
|
|
run_async(
|
|
py,
|
|
async move { service.async_lookup_batch(&requests, now()).await },
|
|
cache_error,
|
|
)
|
|
}
|
|
CacheBinding::PythonCallback(callback) => {
|
|
callback.async_lookup_batch(py, requests, callback_kwargs)
|
|
}
|
|
}
|
|
}
|
|
|
|
#[pyo3(signature = (requests, responses, *, callback_result=None, callback_kwargs=None))]
|
|
fn async_store_batch<'py>(
|
|
&self,
|
|
py: Python<'py>,
|
|
requests: &Bound<'py, PyAny>,
|
|
responses: &Bound<'py, PyAny>,
|
|
callback_result: Option<&Bound<'py, PyAny>>,
|
|
callback_kwargs: Option<&Bound<'py, PyDict>>,
|
|
) -> PyResult<Bound<'py, PyAny>> {
|
|
self.check_process()?;
|
|
match &self.binding {
|
|
CacheBinding::Disabled => ready_none(py),
|
|
CacheBinding::Native(service) => {
|
|
let requests = self::requests(requests)?;
|
|
let responses: Vec<Value> = from_py(responses)?;
|
|
if requests.len() != responses.len() {
|
|
return Err(PyValueError::new_err(
|
|
"batch cache requests and responses must have equal lengths",
|
|
));
|
|
}
|
|
let entries = requests.into_iter().zip(responses).collect();
|
|
service.async_store_batch_py(py, entries)
|
|
}
|
|
CacheBinding::PythonCallback(callback) => {
|
|
callback.async_store_batch(py, callback_result, callback_kwargs)
|
|
}
|
|
}
|
|
}
|
|
|
|
fn async_flush<'py>(&self, py: Python<'py>) -> PyResult<Bound<'py, PyAny>> {
|
|
self.check_process()?;
|
|
match &self.binding {
|
|
CacheBinding::Disabled => ready_none(py),
|
|
CacheBinding::Native(service) => {
|
|
let service = service.clone();
|
|
run_async(py, async move { service.async_flush().await }, cache_error)
|
|
}
|
|
CacheBinding::PythonCallback(callback) => callback.async_flush(py),
|
|
}
|
|
}
|
|
|
|
fn ping<'py>(&self, py: Python<'py>) -> PyResult<Bound<'py, PyAny>> {
|
|
self.check_process()?;
|
|
match &self.binding {
|
|
CacheBinding::Disabled => ready_none(py),
|
|
CacheBinding::Native(service) => {
|
|
let service = service.clone();
|
|
run_async(
|
|
py,
|
|
async move { service.test_connection().await },
|
|
cache_error,
|
|
)
|
|
}
|
|
CacheBinding::PythonCallback(callback) => callback.ping(py),
|
|
}
|
|
}
|
|
|
|
fn __traverse__(&self, visit: PyVisit<'_>) -> Result<(), PyTraverseError> {
|
|
if let CacheBinding::PythonCallback(callback) = &self.binding {
|
|
callback.traverse(&visit)?;
|
|
}
|
|
Ok(())
|
|
}
|
|
}
|