import os import pytesseract from pdf2image import convert_from_path import fitz # PyMuPDF import docx from PIL import Image import io def extract_text_from_pdf(file_path): """ Extract text from PDF using PyMuPDF (fitz). Falls back to OCR if needed. """ try: # First try direct text extraction doc = fitz.open(file_path) text = "" for page in doc: text += page.get_text() # If no text was extracted, try OCR if not text.strip(): return extract_text_with_ocr(file_path) return text except Exception as e: print(f"Error extracting text from PDF: {e}") return extract_text_with_ocr(file_path) def extract_text_with_ocr(file_path): """ Extract text using OCR for scanned documents. """ try: # Convert PDF to images images = convert_from_path(file_path) text = "" # Process each page with OCR for img in images: text += pytesseract.image_to_string(img) return text except Exception as e: print(f"OCR extraction error: {e}") return "" def extract_text_from_image(file_path): """ Extract text from images using OCR. """ try: img = Image.open(file_path) text = pytesseract.image_to_string(img) return text except Exception as e: print(f"Image OCR error: {e}") return "" def extract_text_from_docx(file_path): """ Extract text from DOCX files. """ try: doc = docx.Document(file_path) full_text = [] for para in doc.paragraphs: full_text.append(para.text) return "\n".join(full_text) except Exception as e: print(f"DOCX extraction error: {e}") return "" def extract_text_from_txt(file_path): """ Extract text from plain text files. """ try: with open(file_path, 'r', encoding='utf-8') as file: return file.read() except UnicodeDecodeError: # Try with a different encoding if UTF-8 fails try: with open(file_path, 'r', encoding='latin-1') as file: return file.read() except Exception as e: print(f"Text file reading error: {e}") return "" except Exception as e: print(f"Text file reading error: {e}") return "" def extract_text_from_file(file_path): """ Extract text from a file based on its extension. """ file_ext = os.path.splitext(file_path)[1].lower() if file_ext in ['.pdf']: return extract_text_from_pdf(file_path) elif file_ext in ['.png', '.jpg', '.jpeg']: return extract_text_from_image(file_path) elif file_ext in ['.docx', '.doc']: return extract_text_from_docx(file_path) elif file_ext in ['.txt']: return extract_text_from_txt(file_path) else: return "Unsupported file format" def process_document(file_object, file_path): """ Process an uploaded document, extract text, and prepare for vectorization. """ try: # Extract text based on file type text_content = extract_text_from_file(file_path) # Basic document preprocessing if text_content: # Remove excessive whitespace text_content = ' '.join(text_content.split()) return { "content": text_content, "success": bool(text_content), "error": "" if text_content else "Failed to extract text" } except Exception as e: return { "content": "", "success": False, "error": str(e) }