Download src/ocr/md_loader.py from gitmodelmujtaba/medical-guidelines-kg: direct link, hf CLI and curl.
- Browser
- Download file 2.17 kB
-
https://huggingface.co/spaces/gitmodelmujtaba/medical-guidelines-kg/resolve/main/src/ocr/md_loader.py
- Command line
-
hf download hf://spaces/gitmodelmujtaba/medical-guidelines-kg/src/ocr/md_loader.py
-
curl -L -o md_loader.py https://huggingface.co/spaces/gitmodelmujtaba/medical-guidelines-kg/resolve/main/src/ocr/md_loader.py
2.17 kB
| """ | |
| Markdown Document Loader. | |
| Loads pre-converted Markdown files, parses page boundaries and metadata, | |
| and produces a ParsedDocument object for downstream processing. | |
| """ | |
| import os | |
| import re | |
| from typing import Dict, List, Optional | |
| from src.ocr.base_parser import BaseDocumentParser, ParsedDocument, ParsedPage | |
| class MarkdownDocumentLoader(BaseDocumentParser): | |
| """Loads and normalizes guideline Markdown files.""" | |
| def parse(self, file_path: str) -> ParsedDocument: | |
| if not os.path.exists(file_path): | |
| raise FileNotFoundError(f"Markdown file not found: {file_path}") | |
| with open(file_path, "r", encoding="utf-8") as f: | |
| content = f.read() | |
| # Split into pages by '***' or '---' horizontal rules or page delimiters | |
| raw_pages = re.split(r"\n\s*(?:\*\*\*|---|___)\s*\n", content) | |
| if len(raw_pages) <= 1: | |
| # Fallback: check for page markers like "Page X of Y" or "\d+ of \d+" | |
| raw_pages = [content] | |
| parsed_pages: List[ParsedPage] = [] | |
| for idx, page_text in enumerate(raw_pages, start=1): | |
| # Clean up repetitive footer/header lines from the page text | |
| cleaned_text = self._clean_page_artifacts(page_text) | |
| parsed_pages.append( | |
| ParsedPage( | |
| page_number=idx, | |
| markdown_content=cleaned_text.strip(), | |
| bbox_metadata=[], | |
| ) | |
| ) | |
| return ParsedDocument( | |
| source_path=file_path, | |
| total_pages=len(parsed_pages), | |
| full_markdown=content, | |
| pages=parsed_pages, | |
| metadata={ | |
| "file_name": os.path.basename(file_path), | |
| "file_size": os.path.getsize(file_path), | |
| }, | |
| ) | |
| def _clean_page_artifacts(self, page_text: str) -> str: | |
| # Remove trailing copyright and pagination footers if standard | |
| cleaned = re.sub( | |
| r"\n\s*(?:©\s*Royal College of Obstetricians and Gynaecologists|©\s*RCOG).*$", | |
| "", | |
| page_text, | |
| flags=re.IGNORECASE | re.MULTILINE, | |
| ) | |
| return cleaned | |