mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
* refactor(cache): organize v2 cache as a package * docs: clarify experimental v2 guidance * fix(cache): verify cache-hit accounting and preserve logging metadata * refactor(cache): separate execution facts from host accounting * refactor(rust): build messages routes with named dependencies * wip * fix(cache): preserve facade policy and preflight fallback * refactor(cache): defer shared Python logging changes * test(gateway-inference): allow dead code in shared test helpers Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(cache): key prepared requests and honor facade controls * feat(cache): use Python caches from Rust Messages inference * refactor(cache): separate native and Python cache adapters * refactor(cache): enforce shared composition and adapter boundaries * fix(cache): let Python key delegated Rust Messages entries --------- Co-authored-by: Yujong Lee <yujong@berri.ai> Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
17 lines
616 B
Python
17 lines
616 B
Python
from typing import Final, TypeVar
|
|
|
|
from litellm.router_utils.add_retry_fallback_headers import (
|
|
_add_headers_to_response, # pyright: ignore[reportPrivateUsage] # reuse the proxy's identity-preserving response metadata writer
|
|
get_hidden_params_dict,
|
|
)
|
|
|
|
ResultT: Final = TypeVar("ResultT")
|
|
|
|
|
|
def mark_rust_response(response: ResultT) -> ResultT:
|
|
cache_key: Final = get_hidden_params_dict(response).get("cache_key")
|
|
_add_headers_to_response(
|
|
response,
|
|
{"x-litellm-rust": "true", **({"x-litellm-cache-key": cache_key} if isinstance(cache_key, str) else {})},
|
|
)
|
|
return response
|