Spaces:
Paused
Paused
| """Extract plain-text context from uploaded office/binary documents. | |
| The ML Intern agent runtime is a coding agent that mostly runs on text-only | |
| LLM backends (many of the free OpenRouter models have no vision). So instead | |
| of just handing the raw binary file to the model, we extract the meaningful | |
| text out of it and present it as a compact, readable context block. | |
| Supported formats: | |
| PDF (.pdf) | |
| Images (.png, .jpg, .jpeg, .webp, .bmp, .tiff, .gif) | |
| Excel (.xlsx, .xlsm, .xls) | |
| Word (.docx) | |
| PowerPoint (.pptx) | |
| OpenDocument (.odt, .ods, .odp) (via LibreOffice) | |
| Plain text (.txt, .md, .csv, .json, .jsonl, .rtf, .log) | |
| IMPORTANT: For OCR we rely on pytesseract + a system tesseract binary. | |
| For legacy binary office formats (.xls, .doc, .rtf, .odt/.ods/.odp) we rely | |
| on LibreOffice headless (`libreoffice --headless --convert-to txt`). If those | |
| system binaries are not installed, extraction degrades gracefully: the file is | |
| still uploaded and referenced, but with a warning that the raw content could | |
| not be parsed. | |
| """ | |
| from __future__ import annotations | |
| import io | |
| import os | |
| import re | |
| import shutil | |
| import subprocess | |
| import tempfile | |
| from typing import Optional | |
| # --- Optional heavy imports, guarded -------------------------------- | |
| try: | |
| from PIL import Image | |
| except Exception: # pragma: no cover | |
| Image = None | |
| try: | |
| from pypdf import PdfReader | |
| except Exception: # pragma: no cover | |
| PdfReader = None | |
| try: | |
| import docx # python-docx | |
| except Exception: # pragma: no cover | |
| docx = None | |
| try: | |
| import openpyxl | |
| except Exception: # pragma: no cover | |
| openpyxl = None | |
| try: | |
| import pptx # python-pptx | |
| except Exception: # pragma: no cover | |
| pptx = None | |
| MAX_CHARS = 60_000 # cap extracted text so the context block stays small | |
| _PAGE_BREAK = "\n\n--- PAGE BREAK ---\n\n" | |
| # Map extension -> parser kind | |
| IMAGE_EXTS = {"png", "jpg", "jpeg", "webp", "bmp", "tiff", "tif", "gif"} | |
| TEXT_EXTS = {"txt", "md", "csv", "json", "jsonl", "log", "rtf"} | |
| _LO_TEXT_FORMATS = {".xls", ".doc", ".odt", ".ods", ".odp"} | |
| def _snippet(text: str, limit: int = 120) -> str: | |
| text = " ".join(text.split()) | |
| return text[:limit] + ("…" if len(text) > limit else "") | |
| # --- Individual extractors ------------------------------------------- | |
| def _extract_pdf(data: bytes) -> str: | |
| if PdfReader is None: | |
| raise RuntimeError("pypdf not installed") | |
| reader = PdfReader(io.BytesIO(data)) | |
| pages = [] | |
| for page in reader.pages: | |
| try: | |
| pages.append(page.extract_text() or "") | |
| except Exception: | |
| pages.append("") | |
| return _PAGE_BREAK.join(pages) | |
| def _extract_docx(data: bytes) -> str: | |
| if docx is None: | |
| raise RuntimeError("python-docx not installed") | |
| document = docx.Document(io.BytesIO(data)) | |
| parts = [] | |
| # Tables first (structure is usually important) | |
| for table in document.tables: | |
| for row in table.rows: | |
| cells = [c.text.strip() for c in row.cells] | |
| parts.append(" | ".join(cells)) | |
| parts.append("") | |
| for para in document.paragraphs: | |
| t = para.text.strip() | |
| if t: | |
| parts.append(t) | |
| return "\n".join(parts) | |
| def _extract_xlsx(data: bytes) -> str: | |
| if openpyxl is None: | |
| raise RuntimeError("openpyxl not installed") | |
| wb = openpyxl.load_workbook(io.BytesIO(data), read_only=True, data_only=True) | |
| parts = [] | |
| for ws in wb.worksheets: | |
| parts.append(f"### Sheet: {ws.title}") | |
| for row in ws.iter_rows(values_only=True): | |
| vals = ["" if v is None else str(v) for v in row] | |
| line = " | ".join(vals).strip() | |
| if line: | |
| parts.append(line) | |
| parts.append("") | |
| return "\n".join(parts) | |
| def _extract_pptx(data: bytes) -> str: | |
| if pptx is None: | |
| raise RuntimeError("python-pptx not installed") | |
| prs = pptx.Presentation(io.BytesIO(data)) | |
| parts = [] | |
| for idx, slide in enumerate(prs.slides, start=1): | |
| parts.append(f"### Slide {idx}") | |
| for shape in slide.shapes: | |
| if hasattr(shape, "text") and shape.text and shape.text.strip(): | |
| parts.append(shape.text.strip()) | |
| if shape.has_table if hasattr(shape, "has_table") else False: | |
| for row in shape.table.rows: | |
| parts.append(" | ".join(c.text.strip() for c in row.cells)) | |
| parts.append("") | |
| return "\n".join(parts) | |
| def _extract_image(data: bytes) -> str: | |
| """OCR an image. Returns combined text + a note about the image.""" | |
| if Image is None: | |
| raise RuntimeError("Pillow not installed") | |
| tesseract = shutil.which("tesseract") | |
| if not tesseract: | |
| raise RuntimeError( | |
| "OCR unavailable: tesseract binary not found on this runtime. " | |
| "The image was still uploaded and attached by reference." | |
| ) | |
| try: | |
| import pytesseract | |
| except Exception as exc: # pragma: no cover | |
| raise RuntimeError(f"OCR unavailable: pytesseract import failed: {exc}") | |
| image = Image.open(io.BytesIO(data)) | |
| try: | |
| text = pytesseract.image_to_string(image) | |
| finally: | |
| try: | |
| image.close() | |
| except Exception: | |
| pass | |
| return text | |
| def _extract_rtf(data: bytes) -> str: | |
| """RTF is too messy to hand-parse well; route through LibreOffice if present.""" | |
| return _lo_convert_to_text(data, ".rtf") | |
| def _extract_legacy_office(data: bytes, ext: str) -> str: | |
| """Legacy binary office formats (.xls, .doc) via LibreOffice headless.""" | |
| return _lo_convert_to_text(data, ext) | |
| def _lo_convert_to_text(data: bytes, ext: str) -> str: | |
| """Use LibreOffice headless to convert a document to plain text.""" | |
| soffice = shutil.which("libreoffice") or shutil.which("soffice") | |
| if not soffice: | |
| raise RuntimeError( | |
| "LibreOffice not installed on this runtime; cannot parse this " | |
| "format here. The file was still uploaded and attached by reference." | |
| ) | |
| with tempfile.TemporaryDirectory() as tmp: | |
| in_path = os.path.join(tmp, f"input{ext}") | |
| out_dir = os.path.join(tmp, "out") | |
| os.makedirs(out_dir, exist_ok=True) | |
| with open(in_path, "wb") as f: | |
| f.write(data) | |
| result = subprocess.run( | |
| [soffice, "--headless", "--convert-to", "txt:Text (encoded):UTF8", | |
| "--outdir", out_dir, in_path], | |
| capture_output=True, | |
| timeout=120, | |
| ) | |
| if result.returncode != 0: | |
| raise RuntimeError( | |
| f"LibreOffice failed to convert this file (rc={result.returncode})." | |
| ) | |
| out_files = os.listdir(out_dir) | |
| if not out_files: | |
| raise RuntimeError("LibreOffice produced no output for this file.") | |
| with open(os.path.join(out_dir, out_files[0]), "r", encoding="utf-8", errors="ignore") as f: | |
| return f.read() | |
| # --- Public dispatcher ------------------------------------------------- | |
| def extract_document_text(filename: str, data: bytes) -> tuple[str, str]: | |
| """Extract text from an uploaded document. | |
| Returns: | |
| (extracted_text, warning) — `warning` is a non-empty string when the | |
| extraction is a degraded fallback, otherwise empty. | |
| """ | |
| ext = os.path.splitext(filename or "")[1].lower().lstrip(".") | |
| try: | |
| if ext == "pdf": | |
| text = _extract_pdf(data) | |
| elif ext in IMAGE_EXTS: | |
| text = _extract_image(data) | |
| elif ext in {"xlsx", "xlsm"}: | |
| text = _extract_xlsx(data) | |
| elif ext == "xls": | |
| text = _extract_legacy_office(data, ".xls") | |
| elif ext == "docx": | |
| text = _extract_docx(data) | |
| elif ext == "doc": | |
| text = _extract_legacy_office(data, ".doc") | |
| elif ext == "pptx": | |
| text = _extract_pptx(data) | |
| elif ext in {"odt", "ods", "odp"}: | |
| text = _extract_legacy_office(data, f".{ext}") | |
| elif ext == "rtf": | |
| text = _extract_rtf(data) | |
| elif ext in TEXT_EXTS: | |
| text = data.decode("utf-8", errors="replace") | |
| else: | |
| return "", f"Unsupported format '.{ext}' — file uploaded but no text extracted." | |
| except RuntimeError as e: | |
| return "", str(e) | |
| except Exception as e: # pragma: no cover | |
| return "", f"Failed to extract text from this file: {e}" | |
| cleaned = "\n".join(line.rstrip() for line in text.splitlines()) | |
| cleaned = re.sub(r"\n{4,}", "\n\n\n", cleaned).strip() | |
| if ext in IMAGE_EXTS: | |
| note = ( | |
| f"[The image is {len(data):,} bytes. The text below is OCR output. " | |
| "For layout/diagram questions, refer to the uploaded image file.]\n\n" | |
| ) | |
| cleaned = note + cleaned | |
| if len(cleaned) > MAX_CHARS: | |
| cleaned = cleaned[:MAX_CHARS] + "\n\n... [TRUNCATED — file content exceeds context limit] ..." | |
| return cleaned, "" | |
| # --- Dispatch helper for the upload note ------------------------------- | |
| def format_uploaded_document_context( | |
| *, | |
| filename: str, | |
| stored_filename: str, | |
| repo_id: str, | |
| path_in_repo: str, | |
| hub_url: str, | |
| size_bytes: int, | |
| file_format: str, | |
| extracted_text: str, | |
| warning: str, | |
| ) -> str: | |
| """Build the SYSTEM context note injected into the session for the AI.""" | |
| lines = [ | |
| "[SYSTEM: The user uploaded a document/file for this session.", | |
| "", | |
| "A text extraction was performed below so you can understand the file's", | |
| "content directly. If the text is truncated, empty, or you need the", | |
| "original bytes/format, the original file is preserved on the Hub:", | |
| "", | |
| f"- Repo ID: {repo_id}", | |
| f"- Repo type: dataset", | |
| f"- File in repo: {path_in_repo}", | |
| f"- Original filename: {filename}", | |
| f"- Stored filename: {stored_filename}", | |
| f"- Format: {file_format}", | |
| f"- Size: {size_bytes} bytes", | |
| f"- Hub URL: {hub_url}", | |
| "", | |
| ] | |
| if warning: | |
| lines.extend( | |
| [ | |
| "WARNING (extraction):", | |
| warning, | |
| "", | |
| "You can still try to load/read the original file from the Hub URL", | |
| "above, but understand the plain-text extraction did not succeed.", | |
| "", | |
| ] | |
| ) | |
| lines.extend( | |
| [ | |
| "Extracted text content:", | |
| "```", | |
| extracted_text if extracted_text else "(no text could be extracted)", | |
| "```", | |
| "]", | |
| ] | |
| ) | |
| return "\n".join(lines) | |