chatbot_theme_identifier / extractor.py
Nikpatil's picture
Upload 8 files
407b660 verified
Raw History Blame Contribute Delete
3.69 kB
import os
import pytesseract
from pdf2image import convert_from_path
import fitz # PyMuPDF
import docx
from PIL import Image
import io
def extract_text_from_pdf(file_path):
"""
Extract text from PDF using PyMuPDF (fitz).
Falls back to OCR if needed.
"""
try:
# First try direct text extraction
doc = fitz.open(file_path)
text = ""
for page in doc:
text += page.get_text()
# If no text was extracted, try OCR
if not text.strip():
return extract_text_with_ocr(file_path)
return text
except Exception as e:
print(f"Error extracting text from PDF: {e}")
return extract_text_with_ocr(file_path)
def extract_text_with_ocr(file_path):
"""
Extract text using OCR for scanned documents.
"""
try:
# Convert PDF to images
images = convert_from_path(file_path)
text = ""
# Process each page with OCR
for img in images:
text += pytesseract.image_to_string(img)
return text
except Exception as e:
print(f"OCR extraction error: {e}")
return ""
def extract_text_from_image(file_path):
"""
Extract text from images using OCR.
"""
try:
img = Image.open(file_path)
text = pytesseract.image_to_string(img)
return text
except Exception as e:
print(f"Image OCR error: {e}")
return ""
def extract_text_from_docx(file_path):
"""
Extract text from DOCX files.
"""
try:
doc = docx.Document(file_path)
full_text = []
for para in doc.paragraphs:
full_text.append(para.text)
return "\n".join(full_text)
except Exception as e:
print(f"DOCX extraction error: {e}")
return ""
def extract_text_from_txt(file_path):
"""
Extract text from plain text files.
"""
try:
with open(file_path, 'r', encoding='utf-8') as file:
return file.read()
except UnicodeDecodeError:
# Try with a different encoding if UTF-8 fails
try:
with open(file_path, 'r', encoding='latin-1') as file:
return file.read()
except Exception as e:
print(f"Text file reading error: {e}")
return ""
except Exception as e:
print(f"Text file reading error: {e}")
return ""
def extract_text_from_file(file_path):
"""
Extract text from a file based on its extension.
"""
file_ext = os.path.splitext(file_path)[1].lower()
if file_ext in ['.pdf']:
return extract_text_from_pdf(file_path)
elif file_ext in ['.png', '.jpg', '.jpeg']:
return extract_text_from_image(file_path)
elif file_ext in ['.docx', '.doc']:
return extract_text_from_docx(file_path)
elif file_ext in ['.txt']:
return extract_text_from_txt(file_path)
else:
return "Unsupported file format"
def process_document(file_object, file_path):
"""
Process an uploaded document, extract text, and prepare for vectorization.
"""
try:
# Extract text based on file type
text_content = extract_text_from_file(file_path)
# Basic document preprocessing
if text_content:
# Remove excessive whitespace
text_content = ' '.join(text_content.split())
return {
"content": text_content,
"success": bool(text_content),
"error": "" if text_content else "Failed to extract text"
}
except Exception as e:
return {
"content": "",
"success": False,
"error": str(e)
}