mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
add validation for bucket name etd
This commit is contained in:
parent
f03f1fd671
commit
1aa47fbbad
2 changed files with 33 additions and 2 deletions
9
litellm/rag/ingestion/file_parsers/__init__.py
Normal file
9
litellm/rag/ingestion/file_parsers/__init__.py
Normal file
|
|
@ -0,0 +1,9 @@
|
|||
"""
|
||||
File parsers for RAG ingestion.
|
||||
|
||||
Provides text extraction utilities for various file formats.
|
||||
"""
|
||||
|
||||
from .pdf_parser import extract_text_from_pdf
|
||||
|
||||
__all__ = ["extract_text_from_pdf"]
|
||||
|
|
@ -253,6 +253,20 @@ class S3VectorsRAGIngestion(BaseRAGIngestion, BaseAWSLLM):
|
|||
verbose_logger.debug(
|
||||
f"Ensuring S3 vector bucket exists: {self.vector_bucket_name}"
|
||||
)
|
||||
|
||||
# Validate bucket name (AWS S3 naming rules)
|
||||
if len(self.vector_bucket_name) < 3:
|
||||
raise ValueError(
|
||||
f"Invalid vector_bucket_name '{self.vector_bucket_name}': "
|
||||
f"AWS S3 bucket names must be at least 3 characters long. "
|
||||
f"Please provide a valid bucket name (e.g., 'my-vector-bucket')."
|
||||
)
|
||||
if not self.vector_bucket_name.replace("-", "").replace(".", "").isalnum():
|
||||
raise ValueError(
|
||||
f"Invalid vector_bucket_name '{self.vector_bucket_name}': "
|
||||
f"AWS S3 bucket names can only contain lowercase letters, numbers, hyphens, and periods. "
|
||||
f"Please provide a valid bucket name (e.g., 'my-vector-bucket')."
|
||||
)
|
||||
|
||||
# Try to get bucket info using GetVectorBucket API
|
||||
get_url = f"https://s3vectors.{self.aws_region_name}.api.aws/GetVectorBucket"
|
||||
|
|
@ -455,8 +469,16 @@ class S3VectorsRAGIngestion(BaseRAGIngestion, BaseAWSLLM):
|
|||
await self._ensure_config_initialized()
|
||||
|
||||
if not embeddings or not chunks:
|
||||
verbose_logger.warning("No embeddings or chunks to store")
|
||||
return self.index_name, None
|
||||
error_msg = (
|
||||
"No text content could be extracted from the file for embedding. "
|
||||
"Possible causes:\n"
|
||||
" 1. PDF files require OCR - add 'ocr' config with a vision model (e.g., 'anthropic/claude-3-5-sonnet-20241022')\n"
|
||||
" 2. Binary files cannot be processed - convert to text first\n"
|
||||
" 3. File is empty or contains no extractable text\n"
|
||||
"For PDFs, either enable OCR or use a PDF extraction library to convert to text before ingestion."
|
||||
)
|
||||
verbose_logger.error(error_msg)
|
||||
raise ValueError(error_msg)
|
||||
|
||||
# Prepare vectors for PutVectors API
|
||||
vectors = []
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue