From aba122f5446172ae26407024b1a63c40cbc66799 Mon Sep 17 00:00:00 2001 From: Ishaan Jaffer Date: Tue, 27 Jan 2026 15:58:26 -0800 Subject: [PATCH] init extract_text_from_pdf --- litellm/rag/ingestion/base_ingestion.py | 17 ++++- .../rag/ingestion/file_parsers/pdf_parser.py | 72 +++++++++++++++++++ 2 files changed, 87 insertions(+), 2 deletions(-) create mode 100644 litellm/rag/ingestion/file_parsers/pdf_parser.py diff --git a/litellm/rag/ingestion/base_ingestion.py b/litellm/rag/ingestion/base_ingestion.py index 20059487b43..3daa767188d 100644 --- a/litellm/rag/ingestion/base_ingestion.py +++ b/litellm/rag/ingestion/base_ingestion.py @@ -24,6 +24,7 @@ from litellm.llms.custom_httpx.http_handler import ( get_async_httpx_client, httpxSpecialProvider, ) +from litellm.rag.ingestion.file_parsers import extract_text_from_pdf from litellm.rag.text_splitters import RecursiveCharacterTextSplitter from litellm.types.rag import RAGIngestOptions, RAGIngestResponse @@ -193,11 +194,23 @@ class BaseRAGIngestion(ABC): if text: text_to_chunk = text elif file_content and not ocr_was_used: + # Try UTF-8 decode first try: text_to_chunk = file_content.decode("utf-8") except UnicodeDecodeError: - verbose_logger.debug("Binary file detected, skipping text chunking") - return [] + # Check if it's a PDF and try to extract text + if file_content.startswith(b"%PDF"): + verbose_logger.debug("PDF detected, attempting text extraction") + text_to_chunk = extract_text_from_pdf(file_content) + if not text_to_chunk: + verbose_logger.debug( + "PDF text extraction failed. Install 'pypdf' or 'PyPDF2' for PDF support, " + "or enable OCR with a vision model." + ) + return [] + else: + verbose_logger.debug("Binary file detected, skipping text chunking") + return [] if not text_to_chunk: return [] diff --git a/litellm/rag/ingestion/file_parsers/pdf_parser.py b/litellm/rag/ingestion/file_parsers/pdf_parser.py new file mode 100644 index 00000000000..fda77270df9 --- /dev/null +++ b/litellm/rag/ingestion/file_parsers/pdf_parser.py @@ -0,0 +1,72 @@ +""" +PDF text extraction utilities. + +Provides text extraction from PDF files using pypdf or PyPDF2. +""" + +from typing import Optional + +from litellm._logging import verbose_logger + + +def extract_text_from_pdf(file_content: bytes) -> Optional[str]: + """ + Extract text from PDF using pypdf if available. + + Args: + file_content: Raw PDF bytes + + Returns: + Extracted text or None if extraction fails + """ + try: + # Try pypdf first (most common) + try: + from io import BytesIO + + from pypdf import PdfReader + + pdf_file = BytesIO(file_content) + reader = PdfReader(pdf_file) + + text_parts = [] + for page in reader.pages: + text = page.extract_text() + if text: + text_parts.append(text) + + if text_parts: + extracted_text = "\n\n".join(text_parts) + verbose_logger.debug(f"Extracted {len(extracted_text)} characters from PDF using pypdf") + return extracted_text + + except ImportError: + verbose_logger.debug("pypdf not available, trying PyPDF2") + + # Fallback to PyPDF2 + try: + from io import BytesIO + + from PyPDF2 import PdfReader + + pdf_file = BytesIO(file_content) + reader = PdfReader(pdf_file) + + text_parts = [] + for page in reader.pages: + text = page.extract_text() + if text: + text_parts.append(text) + + if text_parts: + extracted_text = "\n\n".join(text_parts) + verbose_logger.debug(f"Extracted {len(extracted_text)} characters from PDF using PyPDF2") + return extracted_text + + except ImportError: + verbose_logger.debug("PyPDF2 not available, PDF extraction requires OCR or pypdf/PyPDF2 library") + + except Exception as e: + verbose_logger.debug(f"PDF text extraction failed: {e}") + + return None