medical-guidelines-kg / src /ocr /md_loader.py
gitmodelmujtaba's picture
Deploy Medical Guidelines Knowledge Graph Explorer to Hugging Face Spaces
62c74ae
Raw History Blame Contribute Delete
2.17 kB
"""
Markdown Document Loader.
Loads pre-converted Markdown files, parses page boundaries and metadata,
and produces a ParsedDocument object for downstream processing.
"""
import os
import re
from typing import Dict, List, Optional
from src.ocr.base_parser import BaseDocumentParser, ParsedDocument, ParsedPage
class MarkdownDocumentLoader(BaseDocumentParser):
"""Loads and normalizes guideline Markdown files."""
def parse(self, file_path: str) -> ParsedDocument:
if not os.path.exists(file_path):
raise FileNotFoundError(f"Markdown file not found: {file_path}")
with open(file_path, "r", encoding="utf-8") as f:
content = f.read()
# Split into pages by '***' or '---' horizontal rules or page delimiters
raw_pages = re.split(r"\n\s*(?:\*\*\*|---|___)\s*\n", content)
if len(raw_pages) <= 1:
# Fallback: check for page markers like "Page X of Y" or "\d+ of \d+"
raw_pages = [content]
parsed_pages: List[ParsedPage] = []
for idx, page_text in enumerate(raw_pages, start=1):
# Clean up repetitive footer/header lines from the page text
cleaned_text = self._clean_page_artifacts(page_text)
parsed_pages.append(
ParsedPage(
page_number=idx,
markdown_content=cleaned_text.strip(),
bbox_metadata=[],
)
)
return ParsedDocument(
source_path=file_path,
total_pages=len(parsed_pages),
full_markdown=content,
pages=parsed_pages,
metadata={
"file_name": os.path.basename(file_path),
"file_size": os.path.getsize(file_path),
},
)
def _clean_page_artifacts(self, page_text: str) -> str:
# Remove trailing copyright and pagination footers if standard
cleaned = re.sub(
r"\n\s*(?:©\s*Royal College of Obstetricians and Gynaecologists|©\s*RCOG).*$",
"",
page_text,
flags=re.IGNORECASE | re.MULTILINE,
)
return cleaned