""" Markdown Document Loader. Loads pre-converted Markdown files, parses page boundaries and metadata, and produces a ParsedDocument object for downstream processing. """ import os import re from typing import Dict, List, Optional from src.ocr.base_parser import BaseDocumentParser, ParsedDocument, ParsedPage class MarkdownDocumentLoader(BaseDocumentParser): """Loads and normalizes guideline Markdown files.""" def parse(self, file_path: str) -> ParsedDocument: if not os.path.exists(file_path): raise FileNotFoundError(f"Markdown file not found: {file_path}") with open(file_path, "r", encoding="utf-8") as f: content = f.read() # Split into pages by '***' or '---' horizontal rules or page delimiters raw_pages = re.split(r"\n\s*(?:\*\*\*|---|___)\s*\n", content) if len(raw_pages) <= 1: # Fallback: check for page markers like "Page X of Y" or "\d+ of \d+" raw_pages = [content] parsed_pages: List[ParsedPage] = [] for idx, page_text in enumerate(raw_pages, start=1): # Clean up repetitive footer/header lines from the page text cleaned_text = self._clean_page_artifacts(page_text) parsed_pages.append( ParsedPage( page_number=idx, markdown_content=cleaned_text.strip(), bbox_metadata=[], ) ) return ParsedDocument( source_path=file_path, total_pages=len(parsed_pages), full_markdown=content, pages=parsed_pages, metadata={ "file_name": os.path.basename(file_path), "file_size": os.path.getsize(file_path), }, ) def _clean_page_artifacts(self, page_text: str) -> str: # Remove trailing copyright and pagination footers if standard cleaned = re.sub( r"\n\s*(?:©\s*Royal College of Obstetricians and Gynaecologists|©\s*RCOG).*$", "", page_text, flags=re.IGNORECASE | re.MULTILINE, ) return cleaned