""" EaseMed Universal Document Parser - core extraction logic. Extracts medicine/procurement line items from PDF, image, Excel and Word documents. Uses native text/table extraction where possible and falls back to OCR (Tesseract) for scanned PDFs and images. Shared by both entrypoints: - main.py (FastAPI, for a Docker-SDK Space or local/self-hosted server) - app.py (Gradio, for a Gradio-SDK Space that doesn't need Docker access) """ import io import re from typing import Any, Dict, List, Optional, Tuple import pdfplumber try: import pytesseract from PIL import Image import shutil as _shutil if _shutil.which("tesseract") is None: # Common Windows install location when tesseract isn't on PATH # (e.g. installed via winget without a shell restart). _win_path = r"C:\Program Files\Tesseract-OCR\tesseract.exe" import os as _os if _os.path.exists(_win_path): pytesseract.pytesseract.tesseract_cmd = _win_path TESSERACT_AVAILABLE = True except ImportError: TESSERACT_AVAILABLE = False try: from pdf2image import convert_from_bytes PDF2IMAGE_AVAILABLE = True except ImportError: PDF2IMAGE_AVAILABLE = False try: from docx import Document as DocxDocument DOCX_AVAILABLE = True except ImportError: DOCX_AVAILABLE = False try: import pandas as pd PANDAS_AVAILABLE = True except ImportError: PANDAS_AVAILABLE = False # --- CATEGORY DEFINITIONS (therapeutic/product classification) --- CATEGORY_DEFINITIONS = { "Pharmaceuticals & Biologics": [ "tablet", "tab", "capsule", "cap", "syrup", "suspension", "susp", "injection", "inj", "vial", "ampoule", "amp", "drops", "gtt", "inhaler", "vaccine", "insulin", "dose", "drug", "medication", "ointment", "cream", "gel", "lotion", "suppository", "supp", "antibiotic", "antiviral", "analgesic", "anesthetic", "hormone", "steroid", "vitamin", "mineral", "supplement", "lozenge", "patch", "solution", "powder", "elixir", "serum", "antitoxin", "antipyretic", "antihistamine", "antidiabetic", "antihypertensive", "antiseptic", "bronchodilator", "nsaid", "electrolyte", ], "Surgical Products": [ "scalpel", "forceps", "retractor", "clamp", "suture", "stapler", "surgical mesh", "hemostatic", "sealant", "surgical drape", "surgical gown", "laparoscopic", "trocar", "surgical clip", "surgical scissor", ], "Diagnostic Products": [ "diagnostic", "test kit", "glucose test", "reagent", "immunoassay", "rapid test", "urinalysis", "test strip", "lancet", ], "Personal Protective Equipment (PPE)": [ "ppe", "n95", "face shield", "goggles", "protective apron", "isolation gown", "surgical mask", ], "Medical Supplies & Consumables": [ "syringe", "needle", "glove", "disposable", "consumable", "cotton wool", "bandage", "gauze", "medical tape", "cannula", "sachet", "tube", ], } def determine_category(description: str, extra: str = "") -> str: text = f"{description} {extra}".lower() for category, keywords in CATEGORY_DEFINITIONS.items(): for k in keywords: if re.search(r"\b" + re.escape(k) + r"\b", text): return category return "Medical Supplies & Consumables" # --- Column header keyword mapping --- HEADER_KEYWORDS = { "name": ["medicine", "drug", "item", "description", "product", "name", "generic"], "brand": ["brand name", "brand", "trade name", "proprietary name"], "form": ["dosage form", "form", "presentation"], "dosage": ["strength", "dosage", "concentration"], "unit": ["unit of measure", "unit", "uom", "pack"], "quantity": ["quantity", "qty", "required quantity", "qnty"], "category": ["therapeutic category", "category", "class", "type"], } # "Item No." / "Item Number" / "Item #" is a row-serial column, not a data # field. Without this guard, "item" (needed to match headers like plain # "Item" or "Item Description") would make this common header steal the # "name" field before the real "Medicine Name" column is considered, since # each field only maps to its first matching column. ITEM_NUMBER_HEADER_PATTERN = re.compile(r"\bitem\s*(no\.?|number|#)\b", re.IGNORECASE) # Likewise "Brand Name" would otherwise be stolen by the "name" field # (bare "name" is a needed keyword for a plain "Name" column), so a # brand-labeled column is excluded from "name" matching specifically. BRAND_HEADER_PATTERN = re.compile(r"\bbrand\b", re.IGNORECASE) def clean_cell(cell: Optional[str]) -> str: if not cell: return "" return re.sub(r"\s+", " ", cell.replace("\n", " ")).strip() def map_header_columns(header_row: List[str]) -> Dict[str, int]: """Match each column index to a semantic field based on header text. Also checks a whitespace-stripped variant of the header text, since wrapped multi-line headers (e.g. "Quantit" + "y" on separate visual lines) get joined with a space that would otherwise break a keyword match like "quantity". """ mapping: Dict[str, int] = {} for idx, raw in enumerate(header_row): cell = clean_cell(raw).lower() if not cell: continue cell_nospace = cell.replace(" ", "") if ITEM_NUMBER_HEADER_PATTERN.search(cell) or re.search(r"item(no\.?|number|#)", cell_nospace): continue is_brand_column = BRAND_HEADER_PATTERN.search(cell) is not None for field, keywords in HEADER_KEYWORDS.items(): if field in mapping: continue if field == "name" and is_brand_column: continue if any(kw in cell or kw.replace(" ", "") in cell_nospace for kw in keywords): mapping[field] = idx break return mapping def looks_like_header(row: List[str]) -> bool: mapped = map_header_columns(row) return "name" in mapped and "quantity" in mapped # The trailing boundary is a negative lookahead (not \b) because \b never # matches between a non-word character (e.g. "%") and a following space — # "5% " would otherwise silently fail to match and let the regex skip ahead # to a later, wrong number (e.g. matching "20g" instead of "5%"). DOSE_PATTERN = re.compile(r"\b\d+(?:\.\d+)?\s*(?:mg|mcg|ml|g|iu|%)(?![a-zA-Z])", re.IGNORECASE) FORM_KEYWORDS = [ "tablet", "tab", "capsule", "cap", "syrup", "suspension", "susp", "injection", "inj", "vial", "ampoule", "amp", "drops", "inhaler", "ointment", "cream", "gel", "lotion", "suppository", "supp", "solution", "sol", "powder", "elixir", "patch", "lozenge", "sachet", "spray", ] FORM_PATTERN = re.compile(r"\b(" + "|".join(FORM_KEYWORDS) + r")\b", re.IGNORECASE) def extract_embedded_dosage_form(name: str, dosage: str, form: str) -> Tuple[str, str, str]: """When a document has no separate Dosage/Form columns, that info is often embedded directly in the medicine name (e.g. "Paracetamol 500mg Tablet"). Pulls dosage/form out of the name text whenever the caller didn't already get them from a dedicated column, and strips the matched text out of the returned name. """ clean_name = name if not dosage: dose_match = DOSE_PATTERN.search(clean_name) if dose_match: dosage = dose_match.group(0) clean_name = (clean_name[: dose_match.start()] + clean_name[dose_match.end():]).strip() if not form: form_match = FORM_PATTERN.search(clean_name) if form_match: form = form_match.group(0).title() clean_name = (clean_name[: form_match.start()] + clean_name[form_match.end():]).strip() clean_name = re.sub(r"\s{2,}", " ", clean_name).strip(" -,") return clean_name or name, dosage, form def row_to_line_item(row: List[str], column_map: Dict[str, int], index: int) -> Optional[Dict[str, Any]]: def get(field: str) -> str: idx = column_map.get(field) if idx is None or idx >= len(row): return "" return clean_cell(row[idx]) name = get("name") quantity_raw = get("quantity") if not name or not quantity_raw: return None qty_match = re.search(r"\d+", quantity_raw.replace(",", "")) if not qty_match: return None quantity = int(qty_match.group(0)) dosage = get("dosage") form = get("form") if not dosage or not form: name, dosage, form = extract_embedded_dosage_form(name, dosage, form) brand = get("brand") or "Generic" unit = get("unit") or "Unit" category = get("category") or determine_category(name, f"{form} {dosage}") return { "line_item_id": index, "inn_name": name, "brand_name": brand, "dosage": dosage, "form": form, "quantity": quantity, "unit_of_issue": unit, "category": category, } def table_to_line_items(table: List[List[Optional[str]]], start_index: int) -> List[Dict[str, Any]]: """Extract line items from a table, given the table may have a repeated header (multi-page tables often repeat the header row on each page).""" items: List[Dict[str, Any]] = [] column_map: Dict[str, int] = {} next_index = start_index for row in table: cleaned = [clean_cell(c) for c in row] if not any(cleaned): continue if looks_like_header(cleaned): column_map = map_header_columns(cleaned) continue if not column_map: # No header identified yet for this table; skip stray rows # rather than guessing positionally. continue item = row_to_line_item(cleaned, column_map, next_index) if item: items.append(item) next_index += 1 return items # --- Fallback: heuristic line-based parsing for unstructured OCR/plain text --- QTY_PATTERN = re.compile(r"\b(\d{1,6})\b") LEADING_ITEM_NO_PATTERN = re.compile(r"^(\d{1,3})\s+(\S.*)$") SKIP_LINE_PATTERN = re.compile( r"item\s*no|medicine name|dosage form|unit of|therapeutic|document control|" r"date of request|department|requested by|approved by|purpose|procurement details", re.IGNORECASE, ) # Conversational prefixes/suffixes seen in free-text requisition lines # (e.g. "Please supply Adrenaline 1mg Injection, quantity 6") that aren't # part of the medicine name and would otherwise end up stuck to it. LEADING_FILLER_PATTERN = re.compile( r"^(please\s+(?:supply|send|provide)|also\s+need|kindly\s+(?:provide|supply|send)|" r"we\s+need|requesting|send|lastly)\s+", re.IGNORECASE, ) TRAILING_FILLER_PATTERN = re.compile( r"[\s,;:\-]*\b(as\s+well|units?\s+needed|needed|required|quantity|qty)\b.*$", re.IGNORECASE, ) def _line_to_item(text: str, dose_required: bool) -> Optional[Dict[str, Any]]: numbers = QTY_PATTERN.findall(text) if not numbers: return None last_num = numbers[-1] quantity = int(last_num) dose_match = DOSE_PATTERN.search(text) if dose_match: dosage = dose_match.group(0) # Keep everything AROUND the dose (not just before it) since the # form often comes after the dose, e.g. "Amoxicillin 250mg Capsule". remainder = text[: dose_match.start()] + " " + text[dose_match.end():] elif not dose_required: dosage = "" remainder = text else: return None # Drop the trailing quantity number and anything after it (usually # filler like "units needed") rather than only the leading segment. qty_pos = remainder.rfind(last_num) if qty_pos != -1: remainder = remainder[:qty_pos] remainder = LEADING_FILLER_PATTERN.sub("", remainder) remainder = TRAILING_FILLER_PATTERN.sub("", remainder) remainder = remainder.strip(" -.,:\t") if not remainder: return None name, dosage, form = extract_embedded_dosage_form(remainder, dosage, "") name = LEADING_FILLER_PATTERN.sub("", name).strip(" -.,:\t") if not name: return None return { "inn_name": name, "brand_name": "Generic", "dosage": dosage, "form": form, "quantity": quantity, "unit_of_issue": "Unit", "category": determine_category(name), } def parse_text_lines(text: str, start_index: int = 1) -> List[Dict[str, Any]]: items: List[Dict[str, Any]] = [] next_index = start_index expected_item_no = 1 for raw_line in text.splitlines(): line = clean_cell(raw_line) if not line or len(line) < 4: continue if SKIP_LINE_PATTERN.search(line): continue item = None # Prefer a sequential leading item number (1, 2, 3, ...) as the # anchor when present — this survives lines with no explicit dose # (e.g. "9 ORS (Oral Rehydration Salt) Powder WHO Formula Sachet 10"). m = LEADING_ITEM_NO_PATTERN.match(line) if m and int(m.group(1)) == expected_item_no: item = _line_to_item(m.group(2), dose_required=False) if item: expected_item_no += 1 # Fall back to requiring a dose pattern for unnumbered lists. if item is None: item = _line_to_item(line, dose_required=True) if item: item["line_item_id"] = next_index items.append(item) next_index += 1 return items HEADER_HINT_WORDS = { "medicine", "item", "description", "dosage", "form", "strength", "dose", "unit", "quantity", "qty", "category", "type", "therapeutic", "brand", "generic", "product", "name", } def cluster_1d(values: List[float], gap: float) -> List[float]: values = sorted(set(values)) clusters: List[List[float]] = [[values[0]]] for v in values[1:]: if v - clusters[-1][-1] <= gap: clusters[-1].append(v) else: clusters.append([v]) return [sum(c) / len(c) for c in clusters] def nearest_index(value: float, anchors: List[float]) -> int: return min(range(len(anchors)), key=lambda i: abs(anchors[i] - value)) def extract_pdf_by_coordinates(content: bytes) -> List[Dict[str, Any]]: """Reconstructs table rows using word x/y coordinates rather than text order. This correctly handles narrow-column PDFs where cell text wraps across multiple lines and pdfplumber's linear text order interleaves adjacent rows/columns. Rows are anchored on a sequential item-number column (1, 2, 3, ...); every other word is assigned to the nearest such row by vertical position and to a column by horizontal position, which survives wrapped, out-of-order text. """ line_items: List[Dict[str, Any]] = [] expected_item_no = 1 next_line_item_id = 1 with pdfplumber.open(io.BytesIO(content)) as pdf: for page in pdf.pages: words = page.extract_words() if not words: continue header_hits = [ w for w in words if w["text"].strip(".,()").lower() in HEADER_HINT_WORDS ] if not header_hits: continue # A real header row has several hint words clustered within a # small vertical band (accounting for a wrapped, multi-line # header). An isolated single hint word elsewhere on the page # (e.g. "Medicine" in a document title) should not count, so # pick the first vertical cluster containing 2+ hint words. sorted_hits = sorted(header_hits, key=lambda w: w["top"]) hit_clusters: List[List[Dict[str, Any]]] = [[sorted_hits[0]]] for w in sorted_hits[1:]: if w["top"] - hit_clusters[-1][-1]["top"] <= 60: hit_clusters[-1].append(w) else: hit_clusters.append([w]) real_header_cluster = next( (c for c in hit_clusters if len(c) >= 3), hit_clusters[0] ) table_top = real_header_cluster[0]["top"] table_words = [w for w in words if w["top"] >= table_top - 3] if not table_words: continue col_centers = cluster_1d([w["x0"] for w in table_words], gap=12) item_col = 0 # leftmost cluster holds the item-number column anchors: List[float] = [] for w in table_words: if nearest_index(w["x0"], col_centers) != item_col: continue if re.match(r"^\d{1,3}$", w["text"]) and int(w["text"]) == expected_item_no: anchors.append(w["top"]) expected_item_no += 1 if not anchors: continue header_words = [w for w in table_words if w["top"] < anchors[0] - 3] body_words = [w for w in table_words if w["top"] >= anchors[0] - 3] header_cols: Dict[int, List[str]] = {} for w in header_words: c = nearest_index(w["x0"], col_centers) header_cols.setdefault(c, []).append(w["text"]) header_texts = [ "" if c == item_col else " ".join(header_cols.get(c, [])) for c in range(len(col_centers)) ] field_to_col = map_header_columns(header_texts) # A cell's tokens (e.g. "500" and "mg") can land in adjacent # column clusters if their x-gap exceeds the clustering # tolerance, even though they share one (short, compact) # header. Any column with no header text of its own is treated # as a continuation of the nearest preceding mapped column. col_to_field = {idx: field for field, idx in field_to_col.items()} field_to_cols: Dict[str, List[int]] = { field: [idx] for field, idx in field_to_col.items() } mapped_indices = sorted(col_to_field.keys()) for c in range(len(col_centers)): if c == item_col or c in col_to_field or header_texts[c].strip(): continue prev_mapped = max((i for i in mapped_indices if i < c), default=None) if prev_mapped is not None: field_to_cols.setdefault(col_to_field[prev_mapped], []).append(c) rows: Dict[int, Dict[int, List[Tuple[float, float, str]]]] = {} for w in body_words: c = nearest_index(w["x0"], col_centers) if c == item_col: continue r = nearest_index(w["top"], anchors) rows.setdefault(r, {}).setdefault(c, []).append((w["top"], w["x0"], w["text"])) for r_idx in range(len(anchors)): row_cols = rows.get(r_idx, {}) def cell_text(field: str) -> str: col_indices = field_to_cols.get(field) if not col_indices: return "" parts: List[Tuple[float, float, str]] = [] for col_idx in col_indices: parts.extend(row_cols.get(col_idx, [])) ordered = sorted(parts) return re.sub(r"-\s+", "-", " ".join(t for _, _, t in ordered)).strip() name = cell_text("name") quantity_text = cell_text("quantity") qty_match = re.search(r"\d+", quantity_text.replace(",", "")) if not name or not qty_match: continue dosage = cell_text("dosage") form = cell_text("form") if not dosage or not form: name, dosage, form = extract_embedded_dosage_form(name, dosage, form) brand = cell_text("brand") or "Generic" unit = cell_text("unit") or "Unit" category = cell_text("category") or determine_category(name, f"{form} {dosage}") next_line_item_id += 1 line_items.append({ "line_item_id": next_line_item_id - 1, "inn_name": name, "brand_name": brand, "dosage": dosage, "form": form, "quantity": int(qty_match.group(0)), "unit_of_issue": unit, "category": category, }) return line_items # --- Format-specific extraction --- def extract_pdf(content: bytes) -> Tuple[List[Dict[str, Any]], str]: all_text_parts: List[str] = [] table_line_items: List[Dict[str, Any]] = [] next_index = 1 with pdfplumber.open(io.BytesIO(content)) as pdf: for page in pdf.pages: page_text = page.extract_text() or "" all_text_parts.append(page_text) for table in page.extract_tables(): found = table_to_line_items(table, next_index) table_line_items.extend(found) next_index += len(found) full_text = "\n".join(all_text_parts) # Fall back to OCR if there's effectively no extractable text (scanned PDF) if len(full_text.strip()) < 20 and PDF2IMAGE_AVAILABLE and TESSERACT_AVAILABLE: images = convert_from_bytes(content) ocr_text_parts = [pytesseract.image_to_string(img) for img in images] full_text = "\n".join(ocr_text_parts) # Coordinate-based reconstruction handles wrapped/narrow-column tables # (the common case for real-world forms) better than linear text order, # so prefer it whenever the document has a usable item-number column. coordinate_items = extract_pdf_by_coordinates(content) if coordinate_items: return coordinate_items, full_text if table_line_items: return table_line_items, full_text return parse_text_lines(full_text, next_index), full_text def extract_image(content: bytes) -> Tuple[List[Dict[str, Any]], str]: if not TESSERACT_AVAILABLE: raise RuntimeError("OCR engine not available on server") image = Image.open(io.BytesIO(content)) text = pytesseract.image_to_string(image) return parse_text_lines(text), text def extract_docx(content: bytes) -> Tuple[List[Dict[str, Any]], str]: if not DOCX_AVAILABLE: raise RuntimeError("python-docx not available on server") doc = DocxDocument(io.BytesIO(content)) line_items: List[Dict[str, Any]] = [] next_index = 1 for table in doc.tables: rows = [[cell.text for cell in row.cells] for row in table.rows] found = table_to_line_items(rows, next_index) line_items.extend(found) next_index += len(found) paragraph_text = "\n".join(p.text for p in doc.paragraphs) if not line_items: line_items = parse_text_lines(paragraph_text, next_index) return line_items, paragraph_text def extract_excel(content: bytes) -> Tuple[List[Dict[str, Any]], str]: if not PANDAS_AVAILABLE: raise RuntimeError("pandas not available on server") sheets = pd.read_excel(io.BytesIO(content), sheet_name=None, header=None, dtype=str) line_items: List[Dict[str, Any]] = [] next_index = 1 for _, df in sheets.items(): rows = df.fillna("").values.tolist() found = table_to_line_items(rows, next_index) line_items.extend(found) next_index += len(found) return line_items, "" def extract_text_file(content: bytes) -> Tuple[List[Dict[str, Any]], str]: text = content.decode("utf-8", errors="ignore") return parse_text_lines(text), text def guess_title(text: str, filename: str) -> str: for line in text.splitlines(): cleaned = clean_cell(line) if len(cleaned) >= 5: return cleaned[:120] return filename.rsplit(".", 1)[0] class UnsupportedFileType(Exception): pass def parse_document(content: bytes, filename: str) -> Dict[str, Any]: """Dispatches to the right extractor based on file extension and returns the response shape the RFQ upload flow expects: {title, description, sections, line_items, fields, provider}. """ filename = filename or "document" ext = filename.lower().rsplit(".", 1)[-1] if "." in filename else "" try: if ext == "pdf": line_items, text = extract_pdf(content) elif ext in ("jpg", "jpeg", "png", "tiff", "bmp"): line_items, text = extract_image(content) elif ext == "docx": line_items, text = extract_docx(content) elif ext in ("xlsx", "xls"): line_items, text = extract_excel(content) elif ext == "doc": raise UnsupportedFileType( "Legacy .doc format is not supported — please save as .docx and retry" ) elif ext in ("txt", "csv"): line_items, text = extract_text_file(content) else: raise UnsupportedFileType(f"Unsupported file type: .{ext}") except UnsupportedFileType: raise except Exception as e: return { "title": "Error Parsing", "description": str(e), "sections": [], "line_items": [], "fields": [], "provider": "self-hosted", } return { "title": guess_title(text, filename), "description": "", "sections": [], "line_items": line_items, "fields": [], "provider": "self-hosted", }