mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
* refactor(rust): align the cache crates with Python and activate every backend The cache port had drifted: lifecycle and Redis-only operations sat on `BaseCache`, counters were pinned to `f64`, each semantic backend defined its own embedder and prompt handling, and only the in-memory backend could be selected natively. - Split `disconnect` and `test_connection` out of `BaseCache` into optional capabilities, implemented only where the Python class defines them, and give every Redis-only operation its own capability trait. - Decouple counters from the stored value type, so one backend can serve both responses and counters as Python's `RedisCache` does. - Share one `Embedder` and prompt contract in `litellm_cache::semantic`, and make the Redis and Valkey semantic backends generic over their codec. - Port the Python operations that were missing: `async_refresh_ttl`, `async_rpush_and_trim`, `async_set_cache_pipeline_with_ttls`, the DualCache pipeline, sadd, bulk delete and TTL reads, and the semantic-similarity write-back. - Take the HTTP client from the host pool in the GCS, S3 and Azure backends. - Activate all nine backends through the Rust catalog, whose rules all stay `PYTHON_ONLY`, and route the `Cache` facade's storage calls to the native runtime when one is selected. - Give every crate the same layout, move all tests to `tests/` on rstest, and add the shared `litellm-cache-testing` contract suite. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> * fix: freeze native cache request kwargs and batch entries for type discipline Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix: declare semantic lookup methods in the native stub Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(rust): align the cache crates with Python and activate every backend The cache port had drifted: lifecycle and Redis-only operations sat on `BaseCache`, counters were pinned to `f64`, each semantic backend defined its own embedder and prompt handling, and only the in-memory backend could be selected natively. - Split `disconnect` and `test_connection` out of `BaseCache` into optional capabilities, implemented only where the Python class defines them, and give every Redis-only operation its own capability trait. - Decouple counters from the stored value type, so one backend can serve both responses and counters as Python's `RedisCache` does. - Share one `Embedder` and prompt contract in `litellm_cache::semantic`, and make the Redis and Valkey semantic backends generic over their codec. - Port the Python operations that were missing: `async_refresh_ttl`, `async_rpush_and_trim`, `async_set_cache_pipeline_with_ttls`, the DualCache pipeline, sadd, bulk delete and TTL reads, and the semantic-similarity write-back. - Take the HTTP client from the host pool in the GCS, S3 and Azure backends. - Activate all nine backends through the Rust catalog, whose rules all stay `PYTHON_ONLY`, and route the `Cache` facade's storage calls to the native runtime when one is selected. - Give every crate the same layout, move all tests to `tests/` on rstest, and add the shared `litellm-cache-testing` contract suite. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> * fix: freeze native cache request kwargs and batch entries for type discipline Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix: declare semantic lookup methods in the native stub Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(rust): opt the native Messages and tokenizer suites into Rust explicitly #42517 made the Messages, token counter and tokenizer routes Python-only, so tests/test_litellm_rust silently exercised the Python path or failed outright. Each suite now prepends a RUST_OPT_IN rule for its route, keeping native coverage without changing the shipped default. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> * fix(rust): pop one at a time in the Redis 6 lpop pipeline and drop explanatory comments Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: Yujong Lee <yujong@berri.ai> Co-authored-by: Claude Opus 5 <noreply@anthropic.com> Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
136 lines
3.1 KiB
Rust
136 lines
3.1 KiB
Rust
use std::sync::Arc;
|
|
|
|
use litellm_cache::{CacheCodec, CacheScript, Error, ScriptCache};
|
|
|
|
use crate::{
|
|
cache::{RedisCache, namespaced_key},
|
|
connection::Connections,
|
|
};
|
|
|
|
#[derive(Clone, Debug, PartialEq)]
|
|
pub enum RedisArg {
|
|
Bytes(Vec<u8>),
|
|
Integer(i64),
|
|
Float(f64),
|
|
}
|
|
|
|
impl From<&str> for RedisArg {
|
|
fn from(value: &str) -> Self {
|
|
Self::Bytes(value.as_bytes().to_vec())
|
|
}
|
|
}
|
|
|
|
impl From<String> for RedisArg {
|
|
fn from(value: String) -> Self {
|
|
Self::Bytes(value.into_bytes())
|
|
}
|
|
}
|
|
|
|
impl From<Vec<u8>> for RedisArg {
|
|
fn from(value: Vec<u8>) -> Self {
|
|
Self::Bytes(value)
|
|
}
|
|
}
|
|
|
|
impl From<i64> for RedisArg {
|
|
fn from(value: i64) -> Self {
|
|
Self::Integer(value)
|
|
}
|
|
}
|
|
|
|
impl From<f64> for RedisArg {
|
|
fn from(value: f64) -> Self {
|
|
Self::Float(value)
|
|
}
|
|
}
|
|
|
|
impl redis::ToRedisArgs for RedisArg {
|
|
fn write_redis_args<W>(&self, out: &mut W)
|
|
where
|
|
W: ?Sized + redis::RedisWrite,
|
|
{
|
|
match self {
|
|
Self::Bytes(value) => value.write_redis_args(out),
|
|
Self::Integer(value) => value.write_redis_args(out),
|
|
Self::Float(value) => value.write_redis_args(out),
|
|
}
|
|
}
|
|
}
|
|
|
|
pub struct RedisScript<C> {
|
|
connections: Arc<Connections<C>>,
|
|
namespace: Option<String>,
|
|
source: String,
|
|
}
|
|
|
|
impl<C> CacheScript for RedisScript<C>
|
|
where
|
|
C: redis::ConnectionLike + Send + 'static,
|
|
{
|
|
type Argument = RedisArg;
|
|
type Output = redis::Value;
|
|
|
|
async fn invoke(
|
|
&self,
|
|
keys: Vec<String>,
|
|
arguments: Vec<Self::Argument>,
|
|
) -> Result<Self::Output, Error> {
|
|
let keys = keys
|
|
.into_iter()
|
|
.map(|key| namespaced_key(self.namespace.as_deref(), &key))
|
|
.collect::<Vec<_>>();
|
|
let source = self.source.clone();
|
|
Connections::run_blocking(Arc::clone(&self.connections), move |connection| {
|
|
eval(connection, &source, keys, arguments)
|
|
})
|
|
.await
|
|
}
|
|
}
|
|
|
|
impl<S, C> RedisCache<S, C>
|
|
where
|
|
S: CacheCodec,
|
|
C: redis::ConnectionLike + Send + 'static,
|
|
{
|
|
pub async fn async_eval(
|
|
&self,
|
|
script: String,
|
|
keys: Vec<String>,
|
|
arguments: Vec<RedisArg>,
|
|
) -> Result<redis::Value, Error> {
|
|
let keys = self.namespaced_keys(&keys);
|
|
self.run(move |connection| eval(connection, &script, keys, arguments))
|
|
.await
|
|
}
|
|
}
|
|
|
|
fn eval(
|
|
connection: &mut impl redis::ConnectionLike,
|
|
script: &str,
|
|
keys: Vec<String>,
|
|
arguments: Vec<RedisArg>,
|
|
) -> Result<redis::Value, Error> {
|
|
redis::cmd("EVAL")
|
|
.arg(script)
|
|
.arg(keys.len())
|
|
.arg(keys)
|
|
.arg(arguments)
|
|
.query(connection)
|
|
.map_err(|_| Error::Unavailable)
|
|
}
|
|
|
|
impl<S, C> ScriptCache for RedisCache<S, C>
|
|
where
|
|
S: CacheCodec,
|
|
C: redis::ConnectionLike + Send + 'static,
|
|
{
|
|
type Script = RedisScript<C>;
|
|
|
|
fn async_register_script(&self, source: String) -> Self::Script {
|
|
RedisScript {
|
|
connections: Arc::clone(&self.connections),
|
|
namespace: self.namespace.clone(),
|
|
source,
|
|
}
|
|
}
|
|
}
|