Spaces:
Runtime error
Runtime error
Download extractor.py from Nikpatil/chatbot_theme_identifier: direct link, hf CLI and curl.
- Browser
- Download file 3.69 kB
-
https://huggingface.co/spaces/Nikpatil/chatbot_theme_identifier/resolve/main/extractor.py
- Command line
-
hf download hf://spaces/Nikpatil/chatbot_theme_identifier/extractor.py
-
curl -L -o extractor.py https://huggingface.co/spaces/Nikpatil/chatbot_theme_identifier/resolve/main/extractor.py
3.69 kB
| import os | |
| import pytesseract | |
| from pdf2image import convert_from_path | |
| import fitz # PyMuPDF | |
| import docx | |
| from PIL import Image | |
| import io | |
| def extract_text_from_pdf(file_path): | |
| """ | |
| Extract text from PDF using PyMuPDF (fitz). | |
| Falls back to OCR if needed. | |
| """ | |
| try: | |
| # First try direct text extraction | |
| doc = fitz.open(file_path) | |
| text = "" | |
| for page in doc: | |
| text += page.get_text() | |
| # If no text was extracted, try OCR | |
| if not text.strip(): | |
| return extract_text_with_ocr(file_path) | |
| return text | |
| except Exception as e: | |
| print(f"Error extracting text from PDF: {e}") | |
| return extract_text_with_ocr(file_path) | |
| def extract_text_with_ocr(file_path): | |
| """ | |
| Extract text using OCR for scanned documents. | |
| """ | |
| try: | |
| # Convert PDF to images | |
| images = convert_from_path(file_path) | |
| text = "" | |
| # Process each page with OCR | |
| for img in images: | |
| text += pytesseract.image_to_string(img) | |
| return text | |
| except Exception as e: | |
| print(f"OCR extraction error: {e}") | |
| return "" | |
| def extract_text_from_image(file_path): | |
| """ | |
| Extract text from images using OCR. | |
| """ | |
| try: | |
| img = Image.open(file_path) | |
| text = pytesseract.image_to_string(img) | |
| return text | |
| except Exception as e: | |
| print(f"Image OCR error: {e}") | |
| return "" | |
| def extract_text_from_docx(file_path): | |
| """ | |
| Extract text from DOCX files. | |
| """ | |
| try: | |
| doc = docx.Document(file_path) | |
| full_text = [] | |
| for para in doc.paragraphs: | |
| full_text.append(para.text) | |
| return "\n".join(full_text) | |
| except Exception as e: | |
| print(f"DOCX extraction error: {e}") | |
| return "" | |
| def extract_text_from_txt(file_path): | |
| """ | |
| Extract text from plain text files. | |
| """ | |
| try: | |
| with open(file_path, 'r', encoding='utf-8') as file: | |
| return file.read() | |
| except UnicodeDecodeError: | |
| # Try with a different encoding if UTF-8 fails | |
| try: | |
| with open(file_path, 'r', encoding='latin-1') as file: | |
| return file.read() | |
| except Exception as e: | |
| print(f"Text file reading error: {e}") | |
| return "" | |
| except Exception as e: | |
| print(f"Text file reading error: {e}") | |
| return "" | |
| def extract_text_from_file(file_path): | |
| """ | |
| Extract text from a file based on its extension. | |
| """ | |
| file_ext = os.path.splitext(file_path)[1].lower() | |
| if file_ext in ['.pdf']: | |
| return extract_text_from_pdf(file_path) | |
| elif file_ext in ['.png', '.jpg', '.jpeg']: | |
| return extract_text_from_image(file_path) | |
| elif file_ext in ['.docx', '.doc']: | |
| return extract_text_from_docx(file_path) | |
| elif file_ext in ['.txt']: | |
| return extract_text_from_txt(file_path) | |
| else: | |
| return "Unsupported file format" | |
| def process_document(file_object, file_path): | |
| """ | |
| Process an uploaded document, extract text, and prepare for vectorization. | |
| """ | |
| try: | |
| # Extract text based on file type | |
| text_content = extract_text_from_file(file_path) | |
| # Basic document preprocessing | |
| if text_content: | |
| # Remove excessive whitespace | |
| text_content = ' '.join(text_content.split()) | |
| return { | |
| "content": text_content, | |
| "success": bool(text_content), | |
| "error": "" if text_content else "Failed to extract text" | |
| } | |
| except Exception as e: | |
| return { | |
| "content": "", | |
| "success": False, | |
| "error": str(e) | |
| } | |