Download src/ocr/base_parser.py from gitmodelmujtaba/medical-guidelines-kg: direct link, hf CLI and curl.
- Browser
- Download file 743 Bytes
-
https://huggingface.co/spaces/gitmodelmujtaba/medical-guidelines-kg/resolve/main/src/ocr/base_parser.py
- Command line
-
hf download hf://spaces/gitmodelmujtaba/medical-guidelines-kg/src/ocr/base_parser.py
-
curl -L -o base_parser.py https://huggingface.co/spaces/gitmodelmujtaba/medical-guidelines-kg/resolve/main/src/ocr/base_parser.py
743 Bytes
| """ | |
| Base Document Parser interface. | |
| """ | |
| from abc import ABC, abstractmethod | |
| from dataclasses import dataclass | |
| from typing import Dict, List, Optional, Any | |
| class ParsedPage: | |
| page_number: int | |
| markdown_content: str | |
| image_path: Optional[str] = None | |
| bbox_metadata: List[Dict[str, Any]] = None | |
| class ParsedDocument: | |
| source_path: str | |
| total_pages: int | |
| full_markdown: str | |
| pages: List[ParsedPage] | |
| metadata: Dict[str, Any] | |
| class BaseDocumentParser(ABC): | |
| """Abstract base class for converting document files to structured Markdown.""" | |
| def parse(self, file_path: str) -> ParsedDocument: | |
| """Parses a PDF or document into a ParsedDocument.""" | |
| pass | |