From 3470193236507a87aedd7cb251005264a92af8dd Mon Sep 17 00:00:00 2001 From: Daniele Scasciafratte Date: Thu, 14 May 2026 11:53:35 +0200 Subject: [PATCH 1/7] fix(vllm): Prevent side-channel attacks via cache salting --- .../llms/hosted_vllm/chat/transformation.py | 8 ++++- .../hosted_vllm/responses/transformation.py | 32 ++++++++++++++++++- 2 files changed, 38 insertions(+), 2 deletions(-) diff --git a/litellm/llms/hosted_vllm/chat/transformation.py b/litellm/llms/hosted_vllm/chat/transformation.py index 1824314865c..af03ba22954 100644 --- a/litellm/llms/hosted_vllm/chat/transformation.py +++ b/litellm/llms/hosted_vllm/chat/transformation.py @@ -14,7 +14,8 @@ from typing import ( cast, overload, ) - +import base64 +import secrets from litellm.litellm_core_utils.prompt_templates.common_utils import ( _get_image_mime_type_from_url, ) @@ -130,6 +131,11 @@ class HostedVLLMChatConfig(OpenAIGPTConfig): else: non_default_params["reasoning_effort"] = "minimal" + cache_salt = optional_params.get("cache_salt") + if cache_salt is None or cache_salt == "": + cache_salt = base64.b64encode(secrets.token_bytes(16)).decode() + optional_params.setdefault("extra_body", {})["cache_salt"] = cache_salt + return super().map_openai_params( non_default_params, optional_params, model, drop_params ) diff --git a/litellm/llms/hosted_vllm/responses/transformation.py b/litellm/llms/hosted_vllm/responses/transformation.py index 4d44eeda9f9..7cbceeebb21 100644 --- a/litellm/llms/hosted_vllm/responses/transformation.py +++ b/litellm/llms/hosted_vllm/responses/transformation.py @@ -6,12 +6,17 @@ so this config enables direct routing instead of falling back to the chat completions → responses conversion pipeline. """ -from typing import Optional +from typing import Dict, Optional, Union +import base64 +import hashlib +import secrets from litellm.llms.openai.responses.transformation import OpenAIResponsesAPIConfig from litellm.secret_managers.main import get_secret_str from litellm.types.router import GenericLiteLLMParams from litellm.types.utils import LlmProviders +from litellm.types.llms.openai import ResponsesAPIRequestParams +from litellm.types.llms.openai import ResponseInputParam class HostedVLLMResponsesAPIConfig(OpenAIResponsesAPIConfig): @@ -28,6 +33,31 @@ class HostedVLLMResponsesAPIConfig(OpenAIResponsesAPIConfig): def custom_llm_provider(self) -> LlmProviders: return LlmProviders.HOSTED_VLLM + def transform_responses_api_request( + self, + model: str, + input: Union[str, ResponseInputParam], + response_api_optional_request_params: Dict, + litellm_params: GenericLiteLLMParams, + headers: dict, + ) -> Dict: + input = self._validate_input_param(input) + final_request_params = dict( + ResponsesAPIRequestParams(model=model, input=input, **response_api_optional_request_params) + ) + # Generate cache_salt from API key for user-based cache isolation (CVE-2025-46570) + auth_header = headers.get("Authorization", "") + if auth_header.startswith("Bearer "): + api_key = auth_header[7:] + else: + api_key = "" + if api_key: + cache_salt = base64.b64encode(hashlib.sha256(api_key.encode()).digest()).decode() + else: + cache_salt = base64.b64encode(secrets.token_bytes(16)).decode() + final_request_params.setdefault("extra_body", {})["cache_salt"] = cache_salt + return final_request_params + def validate_environment( self, headers: dict, From ead4587e22196937592ed423f9727474e6d5395a Mon Sep 17 00:00:00 2001 From: Daniele Scasciafratte Date: Thu, 14 May 2026 12:04:25 +0200 Subject: [PATCH 2/7] Update litellm/llms/hosted_vllm/chat/transformation.py Co-authored-by: greptile-apps[bot] <165735046+greptile-apps[bot]@users.noreply.github.com> --- litellm/llms/hosted_vllm/chat/transformation.py | 1 + 1 file changed, 1 insertion(+) diff --git a/litellm/llms/hosted_vllm/chat/transformation.py b/litellm/llms/hosted_vllm/chat/transformation.py index af03ba22954..1a02d1a9e8e 100644 --- a/litellm/llms/hosted_vllm/chat/transformation.py +++ b/litellm/llms/hosted_vllm/chat/transformation.py @@ -16,6 +16,7 @@ from typing import ( ) import base64 import secrets + from litellm.litellm_core_utils.prompt_templates.common_utils import ( _get_image_mime_type_from_url, ) From a16b3d74d8005d44e7dfcb69a621e78376d74465 Mon Sep 17 00:00:00 2001 From: Daniele Scasciafratte Date: Thu, 14 May 2026 12:07:53 +0200 Subject: [PATCH 3/7] Update litellm/llms/hosted_vllm/chat/transformation.py Co-authored-by: greptile-apps[bot] <165735046+greptile-apps[bot]@users.noreply.github.com> --- litellm/llms/hosted_vllm/chat/transformation.py | 1 + 1 file changed, 1 insertion(+) diff --git a/litellm/llms/hosted_vllm/chat/transformation.py b/litellm/llms/hosted_vllm/chat/transformation.py index 1a02d1a9e8e..b17dad2a39e 100644 --- a/litellm/llms/hosted_vllm/chat/transformation.py +++ b/litellm/llms/hosted_vllm/chat/transformation.py @@ -15,6 +15,7 @@ from typing import ( overload, ) import base64 +import base64 import secrets from litellm.litellm_core_utils.prompt_templates.common_utils import ( From 0157b201c23791d277da1227587e5304b5bb99f5 Mon Sep 17 00:00:00 2001 From: Daniele Scasciafratte Date: Thu, 14 May 2026 12:08:05 +0200 Subject: [PATCH 4/7] Update litellm/llms/hosted_vllm/responses/transformation.py Co-authored-by: greptile-apps[bot] <165735046+greptile-apps[bot]@users.noreply.github.com> --- litellm/llms/hosted_vllm/responses/transformation.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/litellm/llms/hosted_vllm/responses/transformation.py b/litellm/llms/hosted_vllm/responses/transformation.py index 7cbceeebb21..ff03281d810 100644 --- a/litellm/llms/hosted_vllm/responses/transformation.py +++ b/litellm/llms/hosted_vllm/responses/transformation.py @@ -6,10 +6,10 @@ so this config enables direct routing instead of falling back to the chat completions → responses conversion pipeline. """ -from typing import Dict, Optional, Union import base64 import hashlib import secrets +from typing import Dict, Optional, Union from litellm.llms.openai.responses.transformation import OpenAIResponsesAPIConfig from litellm.secret_managers.main import get_secret_str From 47ce244d6fb6e30bb2620e520c959056cdcc63de Mon Sep 17 00:00:00 2001 From: Daniele Scasciafratte Date: Thu, 14 May 2026 14:28:01 +0200 Subject: [PATCH 5/7] Refactor cache_salt generation in transformation.py Refactor cache_salt generation logic in transform_request method to use API key hash if available. --- .../llms/hosted_vllm/chat/transformation.py | 29 +++++++++++++++---- 1 file changed, 23 insertions(+), 6 deletions(-) diff --git a/litellm/llms/hosted_vllm/chat/transformation.py b/litellm/llms/hosted_vllm/chat/transformation.py index b17dad2a39e..4c9ddbcc77c 100644 --- a/litellm/llms/hosted_vllm/chat/transformation.py +++ b/litellm/llms/hosted_vllm/chat/transformation.py @@ -14,7 +14,7 @@ from typing import ( cast, overload, ) -import base64 +import hashlib import base64 import secrets @@ -133,15 +133,32 @@ class HostedVLLMChatConfig(OpenAIGPTConfig): else: non_default_params["reasoning_effort"] = "minimal" - cache_salt = optional_params.get("cache_salt") - if cache_salt is None or cache_salt == "": - cache_salt = base64.b64encode(secrets.token_bytes(16)).decode() - optional_params.setdefault("extra_body", {})["cache_salt"] = cache_salt - return super().map_openai_params( non_default_params, optional_params, model, drop_params ) + + def transform_request( + self, + model: str, + messages: List[AllMessageValues], + optional_params: dict, + litellm_params: dict, + headers: dict, + ) -> dict: + request = super().transform_request(model, messages, optional_params, litellm_params, headers) + auth_header = headers.get("Authorization", "") + if auth_header.startswith("Bearer "): + api_key = auth_header[7:] + else: + api_key = "" + if api_key: + cache_salt = base64.b64encode(hashlib.sha256(api_key.encode()).digest()).decode() + else: + cache_salt = base64.b64encode(secrets.token_bytes(16)).decode() + request.setdefault("extra_body", {})["cache_salt"] = cache_salt + return request + def _get_openai_compatible_provider_info( self, api_base: Optional[str], api_key: Optional[str] ) -> Tuple[Optional[str], Optional[str]]: From c756acd3e7a9fa87df41fa89c36fd8713eed39e4 Mon Sep 17 00:00:00 2001 From: Daniele Scasciafratte Date: Thu, 14 May 2026 14:31:03 +0200 Subject: [PATCH 6/7] Refactor cache_salt generation logic Refactor cache_salt generation to prioritize caller_id over api_key for user-based cache isolation. --- .../hosted_vllm/responses/transformation.py | 22 +++++++++++-------- 1 file changed, 13 insertions(+), 9 deletions(-) diff --git a/litellm/llms/hosted_vllm/responses/transformation.py b/litellm/llms/hosted_vllm/responses/transformation.py index ff03281d810..f8a02fad076 100644 --- a/litellm/llms/hosted_vllm/responses/transformation.py +++ b/litellm/llms/hosted_vllm/responses/transformation.py @@ -45,17 +45,21 @@ class HostedVLLMResponsesAPIConfig(OpenAIResponsesAPIConfig): final_request_params = dict( ResponsesAPIRequestParams(model=model, input=input, **response_api_optional_request_params) ) - # Generate cache_salt from API key for user-based cache isolation (CVE-2025-46570) - auth_header = headers.get("Authorization", "") - if auth_header.startswith("Bearer "): - api_key = auth_header[7:] - else: - api_key = "" - if api_key: - cache_salt = base64.b64encode(hashlib.sha256(api_key.encode()).digest()).decode() + + if final_request_params.get("cache_salt"): + return final_request_params + + metadata = getattr(litellm_params, "metadata", {}) or {} + caller_id = ( + metadata.get("user_api_key_user_id") + or metadata.get("user_api_key_team_id") + or metadata.get("user_api_key_end_user_id") + ) + if caller_id: + cache_salt = base64.b64encode(hashlib.sha256(caller_id.encode()).digest()).decode() else: cache_salt = base64.b64encode(secrets.token_bytes(16)).decode() - final_request_params.setdefault("extra_body", {})["cache_salt"] = cache_salt + final_request_params["cache_salt"] = cache_salt return final_request_params def validate_environment( From f62059f72023bf6cd2c8d94b9583f596a2cbf56e Mon Sep 17 00:00:00 2001 From: Daniele Scasciafratte Date: Fri, 15 May 2026 10:07:18 +0200 Subject: [PATCH 7/7] Update cache_salt assignment in transform_request Refactor cache_salt handling in request transformation. --- litellm/llms/hosted_vllm/chat/transformation.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/litellm/llms/hosted_vllm/chat/transformation.py b/litellm/llms/hosted_vllm/chat/transformation.py index 4c9ddbcc77c..9b7794e7b1f 100644 --- a/litellm/llms/hosted_vllm/chat/transformation.py +++ b/litellm/llms/hosted_vllm/chat/transformation.py @@ -147,6 +147,9 @@ class HostedVLLMChatConfig(OpenAIGPTConfig): ) -> dict: request = super().transform_request(model, messages, optional_params, litellm_params, headers) + if request.get("cache_salt"): + return request + auth_header = headers.get("Authorization", "") if auth_header.startswith("Bearer "): api_key = auth_header[7:] @@ -156,7 +159,7 @@ class HostedVLLMChatConfig(OpenAIGPTConfig): cache_salt = base64.b64encode(hashlib.sha256(api_key.encode()).digest()).decode() else: cache_salt = base64.b64encode(secrets.token_bytes(16)).decode() - request.setdefault("extra_body", {})["cache_salt"] = cache_salt + request["cache_salt"] = cache_salt return request def _get_openai_compatible_provider_info(