From 1aa47fbbadc2b4eda1833b8f4f1a1da2c7a099c1 Mon Sep 17 00:00:00 2001 From: Ishaan Jaffer Date: Tue, 27 Jan 2026 15:57:39 -0800 Subject: [PATCH] add validation for bucket name etd --- .../rag/ingestion/file_parsers/__init__.py | 9 +++++++ litellm/rag/ingestion/s3_vectors_ingestion.py | 26 +++++++++++++++++-- 2 files changed, 33 insertions(+), 2 deletions(-) create mode 100644 litellm/rag/ingestion/file_parsers/__init__.py diff --git a/litellm/rag/ingestion/file_parsers/__init__.py b/litellm/rag/ingestion/file_parsers/__init__.py new file mode 100644 index 00000000000..5be68cdbb74 --- /dev/null +++ b/litellm/rag/ingestion/file_parsers/__init__.py @@ -0,0 +1,9 @@ +""" +File parsers for RAG ingestion. + +Provides text extraction utilities for various file formats. +""" + +from .pdf_parser import extract_text_from_pdf + +__all__ = ["extract_text_from_pdf"] diff --git a/litellm/rag/ingestion/s3_vectors_ingestion.py b/litellm/rag/ingestion/s3_vectors_ingestion.py index d9abb7aba8d..2a938e47554 100644 --- a/litellm/rag/ingestion/s3_vectors_ingestion.py +++ b/litellm/rag/ingestion/s3_vectors_ingestion.py @@ -253,6 +253,20 @@ class S3VectorsRAGIngestion(BaseRAGIngestion, BaseAWSLLM): verbose_logger.debug( f"Ensuring S3 vector bucket exists: {self.vector_bucket_name}" ) + + # Validate bucket name (AWS S3 naming rules) + if len(self.vector_bucket_name) < 3: + raise ValueError( + f"Invalid vector_bucket_name '{self.vector_bucket_name}': " + f"AWS S3 bucket names must be at least 3 characters long. " + f"Please provide a valid bucket name (e.g., 'my-vector-bucket')." + ) + if not self.vector_bucket_name.replace("-", "").replace(".", "").isalnum(): + raise ValueError( + f"Invalid vector_bucket_name '{self.vector_bucket_name}': " + f"AWS S3 bucket names can only contain lowercase letters, numbers, hyphens, and periods. " + f"Please provide a valid bucket name (e.g., 'my-vector-bucket')." + ) # Try to get bucket info using GetVectorBucket API get_url = f"https://s3vectors.{self.aws_region_name}.api.aws/GetVectorBucket" @@ -455,8 +469,16 @@ class S3VectorsRAGIngestion(BaseRAGIngestion, BaseAWSLLM): await self._ensure_config_initialized() if not embeddings or not chunks: - verbose_logger.warning("No embeddings or chunks to store") - return self.index_name, None + error_msg = ( + "No text content could be extracted from the file for embedding. " + "Possible causes:\n" + " 1. PDF files require OCR - add 'ocr' config with a vision model (e.g., 'anthropic/claude-3-5-sonnet-20241022')\n" + " 2. Binary files cannot be processed - convert to text first\n" + " 3. File is empty or contains no extractable text\n" + "For PDFs, either enable OCR or use a PDF extraction library to convert to text before ingestion." + ) + verbose_logger.error(error_msg) + raise ValueError(error_msg) # Prepare vectors for PutVectors API vectors = []