Spaces:
Running on Zero
Running on Zero
Download parser_core.py from jatindersharmapb/easemed-universal-parser: direct link, hf CLI and curl.
- Browser
- Download file 25.5 kB
-
https://huggingface.co/spaces/jatindersharmapb/easemed-universal-parser/resolve/main/parser_core.py
- Command line
-
hf download hf://spaces/jatindersharmapb/easemed-universal-parser/parser_core.py
-
curl -L -o parser_core.py https://huggingface.co/spaces/jatindersharmapb/easemed-universal-parser/resolve/main/parser_core.py
25.5 kB
| """ | |
| EaseMed Universal Document Parser - core extraction logic. | |
| Extracts medicine/procurement line items from PDF, image, Excel and Word | |
| documents. Uses native text/table extraction where possible and falls back | |
| to OCR (Tesseract) for scanned PDFs and images. | |
| Shared by both entrypoints: | |
| - main.py (FastAPI, for a Docker-SDK Space or local/self-hosted server) | |
| - app.py (Gradio, for a Gradio-SDK Space that doesn't need Docker access) | |
| """ | |
| import io | |
| import re | |
| from typing import Any, Dict, List, Optional, Tuple | |
| import pdfplumber | |
| try: | |
| import pytesseract | |
| from PIL import Image | |
| import shutil as _shutil | |
| if _shutil.which("tesseract") is None: | |
| # Common Windows install location when tesseract isn't on PATH | |
| # (e.g. installed via winget without a shell restart). | |
| _win_path = r"C:\Program Files\Tesseract-OCR\tesseract.exe" | |
| import os as _os | |
| if _os.path.exists(_win_path): | |
| pytesseract.pytesseract.tesseract_cmd = _win_path | |
| TESSERACT_AVAILABLE = True | |
| except ImportError: | |
| TESSERACT_AVAILABLE = False | |
| try: | |
| from pdf2image import convert_from_bytes | |
| PDF2IMAGE_AVAILABLE = True | |
| except ImportError: | |
| PDF2IMAGE_AVAILABLE = False | |
| try: | |
| from docx import Document as DocxDocument | |
| DOCX_AVAILABLE = True | |
| except ImportError: | |
| DOCX_AVAILABLE = False | |
| try: | |
| import pandas as pd | |
| PANDAS_AVAILABLE = True | |
| except ImportError: | |
| PANDAS_AVAILABLE = False | |
| # --- CATEGORY DEFINITIONS (therapeutic/product classification) --- | |
| CATEGORY_DEFINITIONS = { | |
| "Pharmaceuticals & Biologics": [ | |
| "tablet", "tab", "capsule", "cap", "syrup", "suspension", "susp", "injection", "inj", "vial", "ampoule", "amp", | |
| "drops", "gtt", "inhaler", "vaccine", "insulin", "dose", "drug", "medication", "ointment", "cream", "gel", | |
| "lotion", "suppository", "supp", "antibiotic", "antiviral", "analgesic", "anesthetic", "hormone", "steroid", | |
| "vitamin", "mineral", "supplement", "lozenge", "patch", "solution", "powder", "elixir", "serum", "antitoxin", | |
| "antipyretic", "antihistamine", "antidiabetic", "antihypertensive", "antiseptic", "bronchodilator", | |
| "nsaid", "electrolyte", | |
| ], | |
| "Surgical Products": [ | |
| "scalpel", "forceps", "retractor", "clamp", "suture", "stapler", "surgical mesh", "hemostatic", "sealant", | |
| "surgical drape", "surgical gown", "laparoscopic", "trocar", "surgical clip", "surgical scissor", | |
| ], | |
| "Diagnostic Products": [ | |
| "diagnostic", "test kit", "glucose test", "reagent", "immunoassay", "rapid test", "urinalysis", | |
| "test strip", "lancet", | |
| ], | |
| "Personal Protective Equipment (PPE)": [ | |
| "ppe", "n95", "face shield", "goggles", "protective apron", "isolation gown", "surgical mask", | |
| ], | |
| "Medical Supplies & Consumables": [ | |
| "syringe", "needle", "glove", "disposable", "consumable", "cotton wool", "bandage", "gauze", | |
| "medical tape", "cannula", "sachet", "tube", | |
| ], | |
| } | |
| def determine_category(description: str, extra: str = "") -> str: | |
| text = f"{description} {extra}".lower() | |
| for category, keywords in CATEGORY_DEFINITIONS.items(): | |
| for k in keywords: | |
| if re.search(r"\b" + re.escape(k) + r"\b", text): | |
| return category | |
| return "Medical Supplies & Consumables" | |
| # --- Column header keyword mapping --- | |
| HEADER_KEYWORDS = { | |
| "name": ["medicine", "drug", "item", "description", "product", "name", "generic"], | |
| "brand": ["brand name", "brand", "trade name", "proprietary name"], | |
| "form": ["dosage form", "form", "presentation"], | |
| "dosage": ["strength", "dosage", "concentration"], | |
| "unit": ["unit of measure", "unit", "uom", "pack"], | |
| "quantity": ["quantity", "qty", "required quantity", "qnty"], | |
| "category": ["therapeutic category", "category", "class", "type"], | |
| } | |
| # "Item No." / "Item Number" / "Item #" is a row-serial column, not a data | |
| # field. Without this guard, "item" (needed to match headers like plain | |
| # "Item" or "Item Description") would make this common header steal the | |
| # "name" field before the real "Medicine Name" column is considered, since | |
| # each field only maps to its first matching column. | |
| ITEM_NUMBER_HEADER_PATTERN = re.compile(r"\bitem\s*(no\.?|number|#)\b", re.IGNORECASE) | |
| # Likewise "Brand Name" would otherwise be stolen by the "name" field | |
| # (bare "name" is a needed keyword for a plain "Name" column), so a | |
| # brand-labeled column is excluded from "name" matching specifically. | |
| BRAND_HEADER_PATTERN = re.compile(r"\bbrand\b", re.IGNORECASE) | |
| def clean_cell(cell: Optional[str]) -> str: | |
| if not cell: | |
| return "" | |
| return re.sub(r"\s+", " ", cell.replace("\n", " ")).strip() | |
| def map_header_columns(header_row: List[str]) -> Dict[str, int]: | |
| """Match each column index to a semantic field based on header text. | |
| Also checks a whitespace-stripped variant of the header text, since | |
| wrapped multi-line headers (e.g. "Quantit" + "y" on separate visual | |
| lines) get joined with a space that would otherwise break a keyword | |
| match like "quantity". | |
| """ | |
| mapping: Dict[str, int] = {} | |
| for idx, raw in enumerate(header_row): | |
| cell = clean_cell(raw).lower() | |
| if not cell: | |
| continue | |
| cell_nospace = cell.replace(" ", "") | |
| if ITEM_NUMBER_HEADER_PATTERN.search(cell) or re.search(r"item(no\.?|number|#)", cell_nospace): | |
| continue | |
| is_brand_column = BRAND_HEADER_PATTERN.search(cell) is not None | |
| for field, keywords in HEADER_KEYWORDS.items(): | |
| if field in mapping: | |
| continue | |
| if field == "name" and is_brand_column: | |
| continue | |
| if any(kw in cell or kw.replace(" ", "") in cell_nospace for kw in keywords): | |
| mapping[field] = idx | |
| break | |
| return mapping | |
| def looks_like_header(row: List[str]) -> bool: | |
| mapped = map_header_columns(row) | |
| return "name" in mapped and "quantity" in mapped | |
| # The trailing boundary is a negative lookahead (not \b) because \b never | |
| # matches between a non-word character (e.g. "%") and a following space — | |
| # "5% " would otherwise silently fail to match and let the regex skip ahead | |
| # to a later, wrong number (e.g. matching "20g" instead of "5%"). | |
| DOSE_PATTERN = re.compile(r"\b\d+(?:\.\d+)?\s*(?:mg|mcg|ml|g|iu|%)(?![a-zA-Z])", re.IGNORECASE) | |
| FORM_KEYWORDS = [ | |
| "tablet", "tab", "capsule", "cap", "syrup", "suspension", "susp", "injection", "inj", | |
| "vial", "ampoule", "amp", "drops", "inhaler", "ointment", "cream", "gel", "lotion", | |
| "suppository", "supp", "solution", "sol", "powder", "elixir", "patch", "lozenge", | |
| "sachet", "spray", | |
| ] | |
| FORM_PATTERN = re.compile(r"\b(" + "|".join(FORM_KEYWORDS) + r")\b", re.IGNORECASE) | |
| def extract_embedded_dosage_form(name: str, dosage: str, form: str) -> Tuple[str, str, str]: | |
| """When a document has no separate Dosage/Form columns, that info is | |
| often embedded directly in the medicine name (e.g. "Paracetamol 500mg | |
| Tablet"). Pulls dosage/form out of the name text whenever the caller | |
| didn't already get them from a dedicated column, and strips the | |
| matched text out of the returned name. | |
| """ | |
| clean_name = name | |
| if not dosage: | |
| dose_match = DOSE_PATTERN.search(clean_name) | |
| if dose_match: | |
| dosage = dose_match.group(0) | |
| clean_name = (clean_name[: dose_match.start()] + clean_name[dose_match.end():]).strip() | |
| if not form: | |
| form_match = FORM_PATTERN.search(clean_name) | |
| if form_match: | |
| form = form_match.group(0).title() | |
| clean_name = (clean_name[: form_match.start()] + clean_name[form_match.end():]).strip() | |
| clean_name = re.sub(r"\s{2,}", " ", clean_name).strip(" -,") | |
| return clean_name or name, dosage, form | |
| def row_to_line_item(row: List[str], column_map: Dict[str, int], index: int) -> Optional[Dict[str, Any]]: | |
| def get(field: str) -> str: | |
| idx = column_map.get(field) | |
| if idx is None or idx >= len(row): | |
| return "" | |
| return clean_cell(row[idx]) | |
| name = get("name") | |
| quantity_raw = get("quantity") | |
| if not name or not quantity_raw: | |
| return None | |
| qty_match = re.search(r"\d+", quantity_raw.replace(",", "")) | |
| if not qty_match: | |
| return None | |
| quantity = int(qty_match.group(0)) | |
| dosage = get("dosage") | |
| form = get("form") | |
| if not dosage or not form: | |
| name, dosage, form = extract_embedded_dosage_form(name, dosage, form) | |
| brand = get("brand") or "Generic" | |
| unit = get("unit") or "Unit" | |
| category = get("category") or determine_category(name, f"{form} {dosage}") | |
| return { | |
| "line_item_id": index, | |
| "inn_name": name, | |
| "brand_name": brand, | |
| "dosage": dosage, | |
| "form": form, | |
| "quantity": quantity, | |
| "unit_of_issue": unit, | |
| "category": category, | |
| } | |
| def table_to_line_items(table: List[List[Optional[str]]], start_index: int) -> List[Dict[str, Any]]: | |
| """Extract line items from a table, given the table may have a repeated | |
| header (multi-page tables often repeat the header row on each page).""" | |
| items: List[Dict[str, Any]] = [] | |
| column_map: Dict[str, int] = {} | |
| next_index = start_index | |
| for row in table: | |
| cleaned = [clean_cell(c) for c in row] | |
| if not any(cleaned): | |
| continue | |
| if looks_like_header(cleaned): | |
| column_map = map_header_columns(cleaned) | |
| continue | |
| if not column_map: | |
| # No header identified yet for this table; skip stray rows | |
| # rather than guessing positionally. | |
| continue | |
| item = row_to_line_item(cleaned, column_map, next_index) | |
| if item: | |
| items.append(item) | |
| next_index += 1 | |
| return items | |
| # --- Fallback: heuristic line-based parsing for unstructured OCR/plain text --- | |
| QTY_PATTERN = re.compile(r"\b(\d{1,6})\b") | |
| LEADING_ITEM_NO_PATTERN = re.compile(r"^(\d{1,3})\s+(\S.*)$") | |
| SKIP_LINE_PATTERN = re.compile( | |
| r"item\s*no|medicine name|dosage form|unit of|therapeutic|document control|" | |
| r"date of request|department|requested by|approved by|purpose|procurement details", | |
| re.IGNORECASE, | |
| ) | |
| # Conversational prefixes/suffixes seen in free-text requisition lines | |
| # (e.g. "Please supply Adrenaline 1mg Injection, quantity 6") that aren't | |
| # part of the medicine name and would otherwise end up stuck to it. | |
| LEADING_FILLER_PATTERN = re.compile( | |
| r"^(please\s+(?:supply|send|provide)|also\s+need|kindly\s+(?:provide|supply|send)|" | |
| r"we\s+need|requesting|send|lastly)\s+", | |
| re.IGNORECASE, | |
| ) | |
| TRAILING_FILLER_PATTERN = re.compile( | |
| r"[\s,;:\-]*\b(as\s+well|units?\s+needed|needed|required|quantity|qty)\b.*$", | |
| re.IGNORECASE, | |
| ) | |
| def _line_to_item(text: str, dose_required: bool) -> Optional[Dict[str, Any]]: | |
| numbers = QTY_PATTERN.findall(text) | |
| if not numbers: | |
| return None | |
| last_num = numbers[-1] | |
| quantity = int(last_num) | |
| dose_match = DOSE_PATTERN.search(text) | |
| if dose_match: | |
| dosage = dose_match.group(0) | |
| # Keep everything AROUND the dose (not just before it) since the | |
| # form often comes after the dose, e.g. "Amoxicillin 250mg Capsule". | |
| remainder = text[: dose_match.start()] + " " + text[dose_match.end():] | |
| elif not dose_required: | |
| dosage = "" | |
| remainder = text | |
| else: | |
| return None | |
| # Drop the trailing quantity number and anything after it (usually | |
| # filler like "units needed") rather than only the leading segment. | |
| qty_pos = remainder.rfind(last_num) | |
| if qty_pos != -1: | |
| remainder = remainder[:qty_pos] | |
| remainder = LEADING_FILLER_PATTERN.sub("", remainder) | |
| remainder = TRAILING_FILLER_PATTERN.sub("", remainder) | |
| remainder = remainder.strip(" -.,:\t") | |
| if not remainder: | |
| return None | |
| name, dosage, form = extract_embedded_dosage_form(remainder, dosage, "") | |
| name = LEADING_FILLER_PATTERN.sub("", name).strip(" -.,:\t") | |
| if not name: | |
| return None | |
| return { | |
| "inn_name": name, | |
| "brand_name": "Generic", | |
| "dosage": dosage, | |
| "form": form, | |
| "quantity": quantity, | |
| "unit_of_issue": "Unit", | |
| "category": determine_category(name), | |
| } | |
| def parse_text_lines(text: str, start_index: int = 1) -> List[Dict[str, Any]]: | |
| items: List[Dict[str, Any]] = [] | |
| next_index = start_index | |
| expected_item_no = 1 | |
| for raw_line in text.splitlines(): | |
| line = clean_cell(raw_line) | |
| if not line or len(line) < 4: | |
| continue | |
| if SKIP_LINE_PATTERN.search(line): | |
| continue | |
| item = None | |
| # Prefer a sequential leading item number (1, 2, 3, ...) as the | |
| # anchor when present — this survives lines with no explicit dose | |
| # (e.g. "9 ORS (Oral Rehydration Salt) Powder WHO Formula Sachet 10"). | |
| m = LEADING_ITEM_NO_PATTERN.match(line) | |
| if m and int(m.group(1)) == expected_item_no: | |
| item = _line_to_item(m.group(2), dose_required=False) | |
| if item: | |
| expected_item_no += 1 | |
| # Fall back to requiring a dose pattern for unnumbered lists. | |
| if item is None: | |
| item = _line_to_item(line, dose_required=True) | |
| if item: | |
| item["line_item_id"] = next_index | |
| items.append(item) | |
| next_index += 1 | |
| return items | |
| HEADER_HINT_WORDS = { | |
| "medicine", "item", "description", "dosage", "form", "strength", "dose", | |
| "unit", "quantity", "qty", "category", "type", "therapeutic", "brand", | |
| "generic", "product", "name", | |
| } | |
| def cluster_1d(values: List[float], gap: float) -> List[float]: | |
| values = sorted(set(values)) | |
| clusters: List[List[float]] = [[values[0]]] | |
| for v in values[1:]: | |
| if v - clusters[-1][-1] <= gap: | |
| clusters[-1].append(v) | |
| else: | |
| clusters.append([v]) | |
| return [sum(c) / len(c) for c in clusters] | |
| def nearest_index(value: float, anchors: List[float]) -> int: | |
| return min(range(len(anchors)), key=lambda i: abs(anchors[i] - value)) | |
| def extract_pdf_by_coordinates(content: bytes) -> List[Dict[str, Any]]: | |
| """Reconstructs table rows using word x/y coordinates rather than text | |
| order. This correctly handles narrow-column PDFs where cell text wraps | |
| across multiple lines and pdfplumber's linear text order interleaves | |
| adjacent rows/columns. Rows are anchored on a sequential item-number | |
| column (1, 2, 3, ...); every other word is assigned to the nearest such | |
| row by vertical position and to a column by horizontal position, which | |
| survives wrapped, out-of-order text. | |
| """ | |
| line_items: List[Dict[str, Any]] = [] | |
| expected_item_no = 1 | |
| next_line_item_id = 1 | |
| with pdfplumber.open(io.BytesIO(content)) as pdf: | |
| for page in pdf.pages: | |
| words = page.extract_words() | |
| if not words: | |
| continue | |
| header_hits = [ | |
| w for w in words | |
| if w["text"].strip(".,()").lower() in HEADER_HINT_WORDS | |
| ] | |
| if not header_hits: | |
| continue | |
| # A real header row has several hint words clustered within a | |
| # small vertical band (accounting for a wrapped, multi-line | |
| # header). An isolated single hint word elsewhere on the page | |
| # (e.g. "Medicine" in a document title) should not count, so | |
| # pick the first vertical cluster containing 2+ hint words. | |
| sorted_hits = sorted(header_hits, key=lambda w: w["top"]) | |
| hit_clusters: List[List[Dict[str, Any]]] = [[sorted_hits[0]]] | |
| for w in sorted_hits[1:]: | |
| if w["top"] - hit_clusters[-1][-1]["top"] <= 60: | |
| hit_clusters[-1].append(w) | |
| else: | |
| hit_clusters.append([w]) | |
| real_header_cluster = next( | |
| (c for c in hit_clusters if len(c) >= 3), hit_clusters[0] | |
| ) | |
| table_top = real_header_cluster[0]["top"] | |
| table_words = [w for w in words if w["top"] >= table_top - 3] | |
| if not table_words: | |
| continue | |
| col_centers = cluster_1d([w["x0"] for w in table_words], gap=12) | |
| item_col = 0 # leftmost cluster holds the item-number column | |
| anchors: List[float] = [] | |
| for w in table_words: | |
| if nearest_index(w["x0"], col_centers) != item_col: | |
| continue | |
| if re.match(r"^\d{1,3}$", w["text"]) and int(w["text"]) == expected_item_no: | |
| anchors.append(w["top"]) | |
| expected_item_no += 1 | |
| if not anchors: | |
| continue | |
| header_words = [w for w in table_words if w["top"] < anchors[0] - 3] | |
| body_words = [w for w in table_words if w["top"] >= anchors[0] - 3] | |
| header_cols: Dict[int, List[str]] = {} | |
| for w in header_words: | |
| c = nearest_index(w["x0"], col_centers) | |
| header_cols.setdefault(c, []).append(w["text"]) | |
| header_texts = [ | |
| "" if c == item_col else " ".join(header_cols.get(c, [])) | |
| for c in range(len(col_centers)) | |
| ] | |
| field_to_col = map_header_columns(header_texts) | |
| # A cell's tokens (e.g. "500" and "mg") can land in adjacent | |
| # column clusters if their x-gap exceeds the clustering | |
| # tolerance, even though they share one (short, compact) | |
| # header. Any column with no header text of its own is treated | |
| # as a continuation of the nearest preceding mapped column. | |
| col_to_field = {idx: field for field, idx in field_to_col.items()} | |
| field_to_cols: Dict[str, List[int]] = { | |
| field: [idx] for field, idx in field_to_col.items() | |
| } | |
| mapped_indices = sorted(col_to_field.keys()) | |
| for c in range(len(col_centers)): | |
| if c == item_col or c in col_to_field or header_texts[c].strip(): | |
| continue | |
| prev_mapped = max((i for i in mapped_indices if i < c), default=None) | |
| if prev_mapped is not None: | |
| field_to_cols.setdefault(col_to_field[prev_mapped], []).append(c) | |
| rows: Dict[int, Dict[int, List[Tuple[float, float, str]]]] = {} | |
| for w in body_words: | |
| c = nearest_index(w["x0"], col_centers) | |
| if c == item_col: | |
| continue | |
| r = nearest_index(w["top"], anchors) | |
| rows.setdefault(r, {}).setdefault(c, []).append((w["top"], w["x0"], w["text"])) | |
| for r_idx in range(len(anchors)): | |
| row_cols = rows.get(r_idx, {}) | |
| def cell_text(field: str) -> str: | |
| col_indices = field_to_cols.get(field) | |
| if not col_indices: | |
| return "" | |
| parts: List[Tuple[float, float, str]] = [] | |
| for col_idx in col_indices: | |
| parts.extend(row_cols.get(col_idx, [])) | |
| ordered = sorted(parts) | |
| return re.sub(r"-\s+", "-", " ".join(t for _, _, t in ordered)).strip() | |
| name = cell_text("name") | |
| quantity_text = cell_text("quantity") | |
| qty_match = re.search(r"\d+", quantity_text.replace(",", "")) | |
| if not name or not qty_match: | |
| continue | |
| dosage = cell_text("dosage") | |
| form = cell_text("form") | |
| if not dosage or not form: | |
| name, dosage, form = extract_embedded_dosage_form(name, dosage, form) | |
| brand = cell_text("brand") or "Generic" | |
| unit = cell_text("unit") or "Unit" | |
| category = cell_text("category") or determine_category(name, f"{form} {dosage}") | |
| next_line_item_id += 1 | |
| line_items.append({ | |
| "line_item_id": next_line_item_id - 1, | |
| "inn_name": name, | |
| "brand_name": brand, | |
| "dosage": dosage, | |
| "form": form, | |
| "quantity": int(qty_match.group(0)), | |
| "unit_of_issue": unit, | |
| "category": category, | |
| }) | |
| return line_items | |
| # --- Format-specific extraction --- | |
| def extract_pdf(content: bytes) -> Tuple[List[Dict[str, Any]], str]: | |
| all_text_parts: List[str] = [] | |
| table_line_items: List[Dict[str, Any]] = [] | |
| next_index = 1 | |
| with pdfplumber.open(io.BytesIO(content)) as pdf: | |
| for page in pdf.pages: | |
| page_text = page.extract_text() or "" | |
| all_text_parts.append(page_text) | |
| for table in page.extract_tables(): | |
| found = table_to_line_items(table, next_index) | |
| table_line_items.extend(found) | |
| next_index += len(found) | |
| full_text = "\n".join(all_text_parts) | |
| # Fall back to OCR if there's effectively no extractable text (scanned PDF) | |
| if len(full_text.strip()) < 20 and PDF2IMAGE_AVAILABLE and TESSERACT_AVAILABLE: | |
| images = convert_from_bytes(content) | |
| ocr_text_parts = [pytesseract.image_to_string(img) for img in images] | |
| full_text = "\n".join(ocr_text_parts) | |
| # Coordinate-based reconstruction handles wrapped/narrow-column tables | |
| # (the common case for real-world forms) better than linear text order, | |
| # so prefer it whenever the document has a usable item-number column. | |
| coordinate_items = extract_pdf_by_coordinates(content) | |
| if coordinate_items: | |
| return coordinate_items, full_text | |
| if table_line_items: | |
| return table_line_items, full_text | |
| return parse_text_lines(full_text, next_index), full_text | |
| def extract_image(content: bytes) -> Tuple[List[Dict[str, Any]], str]: | |
| if not TESSERACT_AVAILABLE: | |
| raise RuntimeError("OCR engine not available on server") | |
| image = Image.open(io.BytesIO(content)) | |
| text = pytesseract.image_to_string(image) | |
| return parse_text_lines(text), text | |
| def extract_docx(content: bytes) -> Tuple[List[Dict[str, Any]], str]: | |
| if not DOCX_AVAILABLE: | |
| raise RuntimeError("python-docx not available on server") | |
| doc = DocxDocument(io.BytesIO(content)) | |
| line_items: List[Dict[str, Any]] = [] | |
| next_index = 1 | |
| for table in doc.tables: | |
| rows = [[cell.text for cell in row.cells] for row in table.rows] | |
| found = table_to_line_items(rows, next_index) | |
| line_items.extend(found) | |
| next_index += len(found) | |
| paragraph_text = "\n".join(p.text for p in doc.paragraphs) | |
| if not line_items: | |
| line_items = parse_text_lines(paragraph_text, next_index) | |
| return line_items, paragraph_text | |
| def extract_excel(content: bytes) -> Tuple[List[Dict[str, Any]], str]: | |
| if not PANDAS_AVAILABLE: | |
| raise RuntimeError("pandas not available on server") | |
| sheets = pd.read_excel(io.BytesIO(content), sheet_name=None, header=None, dtype=str) | |
| line_items: List[Dict[str, Any]] = [] | |
| next_index = 1 | |
| for _, df in sheets.items(): | |
| rows = df.fillna("").values.tolist() | |
| found = table_to_line_items(rows, next_index) | |
| line_items.extend(found) | |
| next_index += len(found) | |
| return line_items, "" | |
| def extract_text_file(content: bytes) -> Tuple[List[Dict[str, Any]], str]: | |
| text = content.decode("utf-8", errors="ignore") | |
| return parse_text_lines(text), text | |
| def guess_title(text: str, filename: str) -> str: | |
| for line in text.splitlines(): | |
| cleaned = clean_cell(line) | |
| if len(cleaned) >= 5: | |
| return cleaned[:120] | |
| return filename.rsplit(".", 1)[0] | |
| class UnsupportedFileType(Exception): | |
| pass | |
| def parse_document(content: bytes, filename: str) -> Dict[str, Any]: | |
| """Dispatches to the right extractor based on file extension and | |
| returns the response shape the RFQ upload flow expects: | |
| {title, description, sections, line_items, fields, provider}. | |
| """ | |
| filename = filename or "document" | |
| ext = filename.lower().rsplit(".", 1)[-1] if "." in filename else "" | |
| try: | |
| if ext == "pdf": | |
| line_items, text = extract_pdf(content) | |
| elif ext in ("jpg", "jpeg", "png", "tiff", "bmp"): | |
| line_items, text = extract_image(content) | |
| elif ext == "docx": | |
| line_items, text = extract_docx(content) | |
| elif ext in ("xlsx", "xls"): | |
| line_items, text = extract_excel(content) | |
| elif ext == "doc": | |
| raise UnsupportedFileType( | |
| "Legacy .doc format is not supported — please save as .docx and retry" | |
| ) | |
| elif ext in ("txt", "csv"): | |
| line_items, text = extract_text_file(content) | |
| else: | |
| raise UnsupportedFileType(f"Unsupported file type: .{ext}") | |
| except UnsupportedFileType: | |
| raise | |
| except Exception as e: | |
| return { | |
| "title": "Error Parsing", | |
| "description": str(e), | |
| "sections": [], | |
| "line_items": [], | |
| "fields": [], | |
| "provider": "self-hosted", | |
| } | |
| return { | |
| "title": guess_title(text, filename), | |
| "description": "", | |
| "sections": [], | |
| "line_items": line_items, | |
| "fields": [], | |
| "provider": "self-hosted", | |
| } | |