easemed-universal-parser / parser_core.py
jatindersharmapb's picture
Extract Brand Name column instead of hardcoding brand_name='Generic'
b0461cd
Raw History Blame Contribute Delete
25.5 kB
"""
EaseMed Universal Document Parser - core extraction logic.
Extracts medicine/procurement line items from PDF, image, Excel and Word
documents. Uses native text/table extraction where possible and falls back
to OCR (Tesseract) for scanned PDFs and images.
Shared by both entrypoints:
- main.py (FastAPI, for a Docker-SDK Space or local/self-hosted server)
- app.py (Gradio, for a Gradio-SDK Space that doesn't need Docker access)
"""
import io
import re
from typing import Any, Dict, List, Optional, Tuple
import pdfplumber
try:
import pytesseract
from PIL import Image
import shutil as _shutil
if _shutil.which("tesseract") is None:
# Common Windows install location when tesseract isn't on PATH
# (e.g. installed via winget without a shell restart).
_win_path = r"C:\Program Files\Tesseract-OCR\tesseract.exe"
import os as _os
if _os.path.exists(_win_path):
pytesseract.pytesseract.tesseract_cmd = _win_path
TESSERACT_AVAILABLE = True
except ImportError:
TESSERACT_AVAILABLE = False
try:
from pdf2image import convert_from_bytes
PDF2IMAGE_AVAILABLE = True
except ImportError:
PDF2IMAGE_AVAILABLE = False
try:
from docx import Document as DocxDocument
DOCX_AVAILABLE = True
except ImportError:
DOCX_AVAILABLE = False
try:
import pandas as pd
PANDAS_AVAILABLE = True
except ImportError:
PANDAS_AVAILABLE = False
# --- CATEGORY DEFINITIONS (therapeutic/product classification) ---
CATEGORY_DEFINITIONS = {
"Pharmaceuticals & Biologics": [
"tablet", "tab", "capsule", "cap", "syrup", "suspension", "susp", "injection", "inj", "vial", "ampoule", "amp",
"drops", "gtt", "inhaler", "vaccine", "insulin", "dose", "drug", "medication", "ointment", "cream", "gel",
"lotion", "suppository", "supp", "antibiotic", "antiviral", "analgesic", "anesthetic", "hormone", "steroid",
"vitamin", "mineral", "supplement", "lozenge", "patch", "solution", "powder", "elixir", "serum", "antitoxin",
"antipyretic", "antihistamine", "antidiabetic", "antihypertensive", "antiseptic", "bronchodilator",
"nsaid", "electrolyte",
],
"Surgical Products": [
"scalpel", "forceps", "retractor", "clamp", "suture", "stapler", "surgical mesh", "hemostatic", "sealant",
"surgical drape", "surgical gown", "laparoscopic", "trocar", "surgical clip", "surgical scissor",
],
"Diagnostic Products": [
"diagnostic", "test kit", "glucose test", "reagent", "immunoassay", "rapid test", "urinalysis",
"test strip", "lancet",
],
"Personal Protective Equipment (PPE)": [
"ppe", "n95", "face shield", "goggles", "protective apron", "isolation gown", "surgical mask",
],
"Medical Supplies & Consumables": [
"syringe", "needle", "glove", "disposable", "consumable", "cotton wool", "bandage", "gauze",
"medical tape", "cannula", "sachet", "tube",
],
}
def determine_category(description: str, extra: str = "") -> str:
text = f"{description} {extra}".lower()
for category, keywords in CATEGORY_DEFINITIONS.items():
for k in keywords:
if re.search(r"\b" + re.escape(k) + r"\b", text):
return category
return "Medical Supplies & Consumables"
# --- Column header keyword mapping ---
HEADER_KEYWORDS = {
"name": ["medicine", "drug", "item", "description", "product", "name", "generic"],
"brand": ["brand name", "brand", "trade name", "proprietary name"],
"form": ["dosage form", "form", "presentation"],
"dosage": ["strength", "dosage", "concentration"],
"unit": ["unit of measure", "unit", "uom", "pack"],
"quantity": ["quantity", "qty", "required quantity", "qnty"],
"category": ["therapeutic category", "category", "class", "type"],
}
# "Item No." / "Item Number" / "Item #" is a row-serial column, not a data
# field. Without this guard, "item" (needed to match headers like plain
# "Item" or "Item Description") would make this common header steal the
# "name" field before the real "Medicine Name" column is considered, since
# each field only maps to its first matching column.
ITEM_NUMBER_HEADER_PATTERN = re.compile(r"\bitem\s*(no\.?|number|#)\b", re.IGNORECASE)
# Likewise "Brand Name" would otherwise be stolen by the "name" field
# (bare "name" is a needed keyword for a plain "Name" column), so a
# brand-labeled column is excluded from "name" matching specifically.
BRAND_HEADER_PATTERN = re.compile(r"\bbrand\b", re.IGNORECASE)
def clean_cell(cell: Optional[str]) -> str:
if not cell:
return ""
return re.sub(r"\s+", " ", cell.replace("\n", " ")).strip()
def map_header_columns(header_row: List[str]) -> Dict[str, int]:
"""Match each column index to a semantic field based on header text.
Also checks a whitespace-stripped variant of the header text, since
wrapped multi-line headers (e.g. "Quantit" + "y" on separate visual
lines) get joined with a space that would otherwise break a keyword
match like "quantity".
"""
mapping: Dict[str, int] = {}
for idx, raw in enumerate(header_row):
cell = clean_cell(raw).lower()
if not cell:
continue
cell_nospace = cell.replace(" ", "")
if ITEM_NUMBER_HEADER_PATTERN.search(cell) or re.search(r"item(no\.?|number|#)", cell_nospace):
continue
is_brand_column = BRAND_HEADER_PATTERN.search(cell) is not None
for field, keywords in HEADER_KEYWORDS.items():
if field in mapping:
continue
if field == "name" and is_brand_column:
continue
if any(kw in cell or kw.replace(" ", "") in cell_nospace for kw in keywords):
mapping[field] = idx
break
return mapping
def looks_like_header(row: List[str]) -> bool:
mapped = map_header_columns(row)
return "name" in mapped and "quantity" in mapped
# The trailing boundary is a negative lookahead (not \b) because \b never
# matches between a non-word character (e.g. "%") and a following space —
# "5% " would otherwise silently fail to match and let the regex skip ahead
# to a later, wrong number (e.g. matching "20g" instead of "5%").
DOSE_PATTERN = re.compile(r"\b\d+(?:\.\d+)?\s*(?:mg|mcg|ml|g|iu|%)(?![a-zA-Z])", re.IGNORECASE)
FORM_KEYWORDS = [
"tablet", "tab", "capsule", "cap", "syrup", "suspension", "susp", "injection", "inj",
"vial", "ampoule", "amp", "drops", "inhaler", "ointment", "cream", "gel", "lotion",
"suppository", "supp", "solution", "sol", "powder", "elixir", "patch", "lozenge",
"sachet", "spray",
]
FORM_PATTERN = re.compile(r"\b(" + "|".join(FORM_KEYWORDS) + r")\b", re.IGNORECASE)
def extract_embedded_dosage_form(name: str, dosage: str, form: str) -> Tuple[str, str, str]:
"""When a document has no separate Dosage/Form columns, that info is
often embedded directly in the medicine name (e.g. "Paracetamol 500mg
Tablet"). Pulls dosage/form out of the name text whenever the caller
didn't already get them from a dedicated column, and strips the
matched text out of the returned name.
"""
clean_name = name
if not dosage:
dose_match = DOSE_PATTERN.search(clean_name)
if dose_match:
dosage = dose_match.group(0)
clean_name = (clean_name[: dose_match.start()] + clean_name[dose_match.end():]).strip()
if not form:
form_match = FORM_PATTERN.search(clean_name)
if form_match:
form = form_match.group(0).title()
clean_name = (clean_name[: form_match.start()] + clean_name[form_match.end():]).strip()
clean_name = re.sub(r"\s{2,}", " ", clean_name).strip(" -,")
return clean_name or name, dosage, form
def row_to_line_item(row: List[str], column_map: Dict[str, int], index: int) -> Optional[Dict[str, Any]]:
def get(field: str) -> str:
idx = column_map.get(field)
if idx is None or idx >= len(row):
return ""
return clean_cell(row[idx])
name = get("name")
quantity_raw = get("quantity")
if not name or not quantity_raw:
return None
qty_match = re.search(r"\d+", quantity_raw.replace(",", ""))
if not qty_match:
return None
quantity = int(qty_match.group(0))
dosage = get("dosage")
form = get("form")
if not dosage or not form:
name, dosage, form = extract_embedded_dosage_form(name, dosage, form)
brand = get("brand") or "Generic"
unit = get("unit") or "Unit"
category = get("category") or determine_category(name, f"{form} {dosage}")
return {
"line_item_id": index,
"inn_name": name,
"brand_name": brand,
"dosage": dosage,
"form": form,
"quantity": quantity,
"unit_of_issue": unit,
"category": category,
}
def table_to_line_items(table: List[List[Optional[str]]], start_index: int) -> List[Dict[str, Any]]:
"""Extract line items from a table, given the table may have a repeated
header (multi-page tables often repeat the header row on each page)."""
items: List[Dict[str, Any]] = []
column_map: Dict[str, int] = {}
next_index = start_index
for row in table:
cleaned = [clean_cell(c) for c in row]
if not any(cleaned):
continue
if looks_like_header(cleaned):
column_map = map_header_columns(cleaned)
continue
if not column_map:
# No header identified yet for this table; skip stray rows
# rather than guessing positionally.
continue
item = row_to_line_item(cleaned, column_map, next_index)
if item:
items.append(item)
next_index += 1
return items
# --- Fallback: heuristic line-based parsing for unstructured OCR/plain text ---
QTY_PATTERN = re.compile(r"\b(\d{1,6})\b")
LEADING_ITEM_NO_PATTERN = re.compile(r"^(\d{1,3})\s+(\S.*)$")
SKIP_LINE_PATTERN = re.compile(
r"item\s*no|medicine name|dosage form|unit of|therapeutic|document control|"
r"date of request|department|requested by|approved by|purpose|procurement details",
re.IGNORECASE,
)
# Conversational prefixes/suffixes seen in free-text requisition lines
# (e.g. "Please supply Adrenaline 1mg Injection, quantity 6") that aren't
# part of the medicine name and would otherwise end up stuck to it.
LEADING_FILLER_PATTERN = re.compile(
r"^(please\s+(?:supply|send|provide)|also\s+need|kindly\s+(?:provide|supply|send)|"
r"we\s+need|requesting|send|lastly)\s+",
re.IGNORECASE,
)
TRAILING_FILLER_PATTERN = re.compile(
r"[\s,;:\-]*\b(as\s+well|units?\s+needed|needed|required|quantity|qty)\b.*$",
re.IGNORECASE,
)
def _line_to_item(text: str, dose_required: bool) -> Optional[Dict[str, Any]]:
numbers = QTY_PATTERN.findall(text)
if not numbers:
return None
last_num = numbers[-1]
quantity = int(last_num)
dose_match = DOSE_PATTERN.search(text)
if dose_match:
dosage = dose_match.group(0)
# Keep everything AROUND the dose (not just before it) since the
# form often comes after the dose, e.g. "Amoxicillin 250mg Capsule".
remainder = text[: dose_match.start()] + " " + text[dose_match.end():]
elif not dose_required:
dosage = ""
remainder = text
else:
return None
# Drop the trailing quantity number and anything after it (usually
# filler like "units needed") rather than only the leading segment.
qty_pos = remainder.rfind(last_num)
if qty_pos != -1:
remainder = remainder[:qty_pos]
remainder = LEADING_FILLER_PATTERN.sub("", remainder)
remainder = TRAILING_FILLER_PATTERN.sub("", remainder)
remainder = remainder.strip(" -.,:\t")
if not remainder:
return None
name, dosage, form = extract_embedded_dosage_form(remainder, dosage, "")
name = LEADING_FILLER_PATTERN.sub("", name).strip(" -.,:\t")
if not name:
return None
return {
"inn_name": name,
"brand_name": "Generic",
"dosage": dosage,
"form": form,
"quantity": quantity,
"unit_of_issue": "Unit",
"category": determine_category(name),
}
def parse_text_lines(text: str, start_index: int = 1) -> List[Dict[str, Any]]:
items: List[Dict[str, Any]] = []
next_index = start_index
expected_item_no = 1
for raw_line in text.splitlines():
line = clean_cell(raw_line)
if not line or len(line) < 4:
continue
if SKIP_LINE_PATTERN.search(line):
continue
item = None
# Prefer a sequential leading item number (1, 2, 3, ...) as the
# anchor when present — this survives lines with no explicit dose
# (e.g. "9 ORS (Oral Rehydration Salt) Powder WHO Formula Sachet 10").
m = LEADING_ITEM_NO_PATTERN.match(line)
if m and int(m.group(1)) == expected_item_no:
item = _line_to_item(m.group(2), dose_required=False)
if item:
expected_item_no += 1
# Fall back to requiring a dose pattern for unnumbered lists.
if item is None:
item = _line_to_item(line, dose_required=True)
if item:
item["line_item_id"] = next_index
items.append(item)
next_index += 1
return items
HEADER_HINT_WORDS = {
"medicine", "item", "description", "dosage", "form", "strength", "dose",
"unit", "quantity", "qty", "category", "type", "therapeutic", "brand",
"generic", "product", "name",
}
def cluster_1d(values: List[float], gap: float) -> List[float]:
values = sorted(set(values))
clusters: List[List[float]] = [[values[0]]]
for v in values[1:]:
if v - clusters[-1][-1] <= gap:
clusters[-1].append(v)
else:
clusters.append([v])
return [sum(c) / len(c) for c in clusters]
def nearest_index(value: float, anchors: List[float]) -> int:
return min(range(len(anchors)), key=lambda i: abs(anchors[i] - value))
def extract_pdf_by_coordinates(content: bytes) -> List[Dict[str, Any]]:
"""Reconstructs table rows using word x/y coordinates rather than text
order. This correctly handles narrow-column PDFs where cell text wraps
across multiple lines and pdfplumber's linear text order interleaves
adjacent rows/columns. Rows are anchored on a sequential item-number
column (1, 2, 3, ...); every other word is assigned to the nearest such
row by vertical position and to a column by horizontal position, which
survives wrapped, out-of-order text.
"""
line_items: List[Dict[str, Any]] = []
expected_item_no = 1
next_line_item_id = 1
with pdfplumber.open(io.BytesIO(content)) as pdf:
for page in pdf.pages:
words = page.extract_words()
if not words:
continue
header_hits = [
w for w in words
if w["text"].strip(".,()").lower() in HEADER_HINT_WORDS
]
if not header_hits:
continue
# A real header row has several hint words clustered within a
# small vertical band (accounting for a wrapped, multi-line
# header). An isolated single hint word elsewhere on the page
# (e.g. "Medicine" in a document title) should not count, so
# pick the first vertical cluster containing 2+ hint words.
sorted_hits = sorted(header_hits, key=lambda w: w["top"])
hit_clusters: List[List[Dict[str, Any]]] = [[sorted_hits[0]]]
for w in sorted_hits[1:]:
if w["top"] - hit_clusters[-1][-1]["top"] <= 60:
hit_clusters[-1].append(w)
else:
hit_clusters.append([w])
real_header_cluster = next(
(c for c in hit_clusters if len(c) >= 3), hit_clusters[0]
)
table_top = real_header_cluster[0]["top"]
table_words = [w for w in words if w["top"] >= table_top - 3]
if not table_words:
continue
col_centers = cluster_1d([w["x0"] for w in table_words], gap=12)
item_col = 0 # leftmost cluster holds the item-number column
anchors: List[float] = []
for w in table_words:
if nearest_index(w["x0"], col_centers) != item_col:
continue
if re.match(r"^\d{1,3}$", w["text"]) and int(w["text"]) == expected_item_no:
anchors.append(w["top"])
expected_item_no += 1
if not anchors:
continue
header_words = [w for w in table_words if w["top"] < anchors[0] - 3]
body_words = [w for w in table_words if w["top"] >= anchors[0] - 3]
header_cols: Dict[int, List[str]] = {}
for w in header_words:
c = nearest_index(w["x0"], col_centers)
header_cols.setdefault(c, []).append(w["text"])
header_texts = [
"" if c == item_col else " ".join(header_cols.get(c, []))
for c in range(len(col_centers))
]
field_to_col = map_header_columns(header_texts)
# A cell's tokens (e.g. "500" and "mg") can land in adjacent
# column clusters if their x-gap exceeds the clustering
# tolerance, even though they share one (short, compact)
# header. Any column with no header text of its own is treated
# as a continuation of the nearest preceding mapped column.
col_to_field = {idx: field for field, idx in field_to_col.items()}
field_to_cols: Dict[str, List[int]] = {
field: [idx] for field, idx in field_to_col.items()
}
mapped_indices = sorted(col_to_field.keys())
for c in range(len(col_centers)):
if c == item_col or c in col_to_field or header_texts[c].strip():
continue
prev_mapped = max((i for i in mapped_indices if i < c), default=None)
if prev_mapped is not None:
field_to_cols.setdefault(col_to_field[prev_mapped], []).append(c)
rows: Dict[int, Dict[int, List[Tuple[float, float, str]]]] = {}
for w in body_words:
c = nearest_index(w["x0"], col_centers)
if c == item_col:
continue
r = nearest_index(w["top"], anchors)
rows.setdefault(r, {}).setdefault(c, []).append((w["top"], w["x0"], w["text"]))
for r_idx in range(len(anchors)):
row_cols = rows.get(r_idx, {})
def cell_text(field: str) -> str:
col_indices = field_to_cols.get(field)
if not col_indices:
return ""
parts: List[Tuple[float, float, str]] = []
for col_idx in col_indices:
parts.extend(row_cols.get(col_idx, []))
ordered = sorted(parts)
return re.sub(r"-\s+", "-", " ".join(t for _, _, t in ordered)).strip()
name = cell_text("name")
quantity_text = cell_text("quantity")
qty_match = re.search(r"\d+", quantity_text.replace(",", ""))
if not name or not qty_match:
continue
dosage = cell_text("dosage")
form = cell_text("form")
if not dosage or not form:
name, dosage, form = extract_embedded_dosage_form(name, dosage, form)
brand = cell_text("brand") or "Generic"
unit = cell_text("unit") or "Unit"
category = cell_text("category") or determine_category(name, f"{form} {dosage}")
next_line_item_id += 1
line_items.append({
"line_item_id": next_line_item_id - 1,
"inn_name": name,
"brand_name": brand,
"dosage": dosage,
"form": form,
"quantity": int(qty_match.group(0)),
"unit_of_issue": unit,
"category": category,
})
return line_items
# --- Format-specific extraction ---
def extract_pdf(content: bytes) -> Tuple[List[Dict[str, Any]], str]:
all_text_parts: List[str] = []
table_line_items: List[Dict[str, Any]] = []
next_index = 1
with pdfplumber.open(io.BytesIO(content)) as pdf:
for page in pdf.pages:
page_text = page.extract_text() or ""
all_text_parts.append(page_text)
for table in page.extract_tables():
found = table_to_line_items(table, next_index)
table_line_items.extend(found)
next_index += len(found)
full_text = "\n".join(all_text_parts)
# Fall back to OCR if there's effectively no extractable text (scanned PDF)
if len(full_text.strip()) < 20 and PDF2IMAGE_AVAILABLE and TESSERACT_AVAILABLE:
images = convert_from_bytes(content)
ocr_text_parts = [pytesseract.image_to_string(img) for img in images]
full_text = "\n".join(ocr_text_parts)
# Coordinate-based reconstruction handles wrapped/narrow-column tables
# (the common case for real-world forms) better than linear text order,
# so prefer it whenever the document has a usable item-number column.
coordinate_items = extract_pdf_by_coordinates(content)
if coordinate_items:
return coordinate_items, full_text
if table_line_items:
return table_line_items, full_text
return parse_text_lines(full_text, next_index), full_text
def extract_image(content: bytes) -> Tuple[List[Dict[str, Any]], str]:
if not TESSERACT_AVAILABLE:
raise RuntimeError("OCR engine not available on server")
image = Image.open(io.BytesIO(content))
text = pytesseract.image_to_string(image)
return parse_text_lines(text), text
def extract_docx(content: bytes) -> Tuple[List[Dict[str, Any]], str]:
if not DOCX_AVAILABLE:
raise RuntimeError("python-docx not available on server")
doc = DocxDocument(io.BytesIO(content))
line_items: List[Dict[str, Any]] = []
next_index = 1
for table in doc.tables:
rows = [[cell.text for cell in row.cells] for row in table.rows]
found = table_to_line_items(rows, next_index)
line_items.extend(found)
next_index += len(found)
paragraph_text = "\n".join(p.text for p in doc.paragraphs)
if not line_items:
line_items = parse_text_lines(paragraph_text, next_index)
return line_items, paragraph_text
def extract_excel(content: bytes) -> Tuple[List[Dict[str, Any]], str]:
if not PANDAS_AVAILABLE:
raise RuntimeError("pandas not available on server")
sheets = pd.read_excel(io.BytesIO(content), sheet_name=None, header=None, dtype=str)
line_items: List[Dict[str, Any]] = []
next_index = 1
for _, df in sheets.items():
rows = df.fillna("").values.tolist()
found = table_to_line_items(rows, next_index)
line_items.extend(found)
next_index += len(found)
return line_items, ""
def extract_text_file(content: bytes) -> Tuple[List[Dict[str, Any]], str]:
text = content.decode("utf-8", errors="ignore")
return parse_text_lines(text), text
def guess_title(text: str, filename: str) -> str:
for line in text.splitlines():
cleaned = clean_cell(line)
if len(cleaned) >= 5:
return cleaned[:120]
return filename.rsplit(".", 1)[0]
class UnsupportedFileType(Exception):
pass
def parse_document(content: bytes, filename: str) -> Dict[str, Any]:
"""Dispatches to the right extractor based on file extension and
returns the response shape the RFQ upload flow expects:
{title, description, sections, line_items, fields, provider}.
"""
filename = filename or "document"
ext = filename.lower().rsplit(".", 1)[-1] if "." in filename else ""
try:
if ext == "pdf":
line_items, text = extract_pdf(content)
elif ext in ("jpg", "jpeg", "png", "tiff", "bmp"):
line_items, text = extract_image(content)
elif ext == "docx":
line_items, text = extract_docx(content)
elif ext in ("xlsx", "xls"):
line_items, text = extract_excel(content)
elif ext == "doc":
raise UnsupportedFileType(
"Legacy .doc format is not supported — please save as .docx and retry"
)
elif ext in ("txt", "csv"):
line_items, text = extract_text_file(content)
else:
raise UnsupportedFileType(f"Unsupported file type: .{ext}")
except UnsupportedFileType:
raise
except Exception as e:
return {
"title": "Error Parsing",
"description": str(e),
"sections": [],
"line_items": [],
"fields": [],
"provider": "self-hosted",
}
return {
"title": guess_title(text, filename),
"description": "",
"sections": [],
"line_items": line_items,
"fields": [],
"provider": "self-hosted",
}