init extract_text_from_pdf

This commit is contained in:
Ishaan Jaffer 2026-01-27 15:58:26 -08:00
parent 71d1b82ad9
commit aba122f544
2 changed files with 87 additions and 2 deletions

View file

@ -24,6 +24,7 @@ from litellm.llms.custom_httpx.http_handler import (
get_async_httpx_client,
httpxSpecialProvider,
)
from litellm.rag.ingestion.file_parsers import extract_text_from_pdf
from litellm.rag.text_splitters import RecursiveCharacterTextSplitter
from litellm.types.rag import RAGIngestOptions, RAGIngestResponse
@ -193,11 +194,23 @@ class BaseRAGIngestion(ABC):
if text:
text_to_chunk = text
elif file_content and not ocr_was_used:
# Try UTF-8 decode first
try:
text_to_chunk = file_content.decode("utf-8")
except UnicodeDecodeError:
verbose_logger.debug("Binary file detected, skipping text chunking")
return []
# Check if it's a PDF and try to extract text
if file_content.startswith(b"%PDF"):
verbose_logger.debug("PDF detected, attempting text extraction")
text_to_chunk = extract_text_from_pdf(file_content)
if not text_to_chunk:
verbose_logger.debug(
"PDF text extraction failed. Install 'pypdf' or 'PyPDF2' for PDF support, "
"or enable OCR with a vision model."
)
return []
else:
verbose_logger.debug("Binary file detected, skipping text chunking")
return []
if not text_to_chunk:
return []

View file

@ -0,0 +1,72 @@
"""
PDF text extraction utilities.
Provides text extraction from PDF files using pypdf or PyPDF2.
"""
from typing import Optional
from litellm._logging import verbose_logger
def extract_text_from_pdf(file_content: bytes) -> Optional[str]:
"""
Extract text from PDF using pypdf if available.
Args:
file_content: Raw PDF bytes
Returns:
Extracted text or None if extraction fails
"""
try:
# Try pypdf first (most common)
try:
from io import BytesIO
from pypdf import PdfReader
pdf_file = BytesIO(file_content)
reader = PdfReader(pdf_file)
text_parts = []
for page in reader.pages:
text = page.extract_text()
if text:
text_parts.append(text)
if text_parts:
extracted_text = "\n\n".join(text_parts)
verbose_logger.debug(f"Extracted {len(extracted_text)} characters from PDF using pypdf")
return extracted_text
except ImportError:
verbose_logger.debug("pypdf not available, trying PyPDF2")
# Fallback to PyPDF2
try:
from io import BytesIO
from PyPDF2 import PdfReader
pdf_file = BytesIO(file_content)
reader = PdfReader(pdf_file)
text_parts = []
for page in reader.pages:
text = page.extract_text()
if text:
text_parts.append(text)
if text_parts:
extracted_text = "\n\n".join(text_parts)
verbose_logger.debug(f"Extracted {len(extracted_text)} characters from PDF using PyPDF2")
return extracted_text
except ImportError:
verbose_logger.debug("PyPDF2 not available, PDF extraction requires OCR or pypdf/PyPDF2 library")
except Exception as e:
verbose_logger.debug(f"PDF text extraction failed: {e}")
return None