Spaces:
Running on Zero
Running on Zero
Download src/gcmd_classifier/datasets/extraction.py from igerasimov/GCMD_Keyword_Classifier_MVP: direct link, hf CLI and curl.
- Browser
- Download file 19.5 kB
-
https://huggingface.co/spaces/igerasimov/GCMD_Keyword_Classifier_MVP/resolve/main/src/gcmd_classifier/datasets/extraction.py
- Command line
-
hf download hf://spaces/igerasimov/GCMD_Keyword_Classifier_MVP/src/gcmd_classifier/datasets/extraction.py
-
curl -L -o extraction.py https://huggingface.co/spaces/igerasimov/GCMD_Keyword_Classifier_MVP/resolve/main/src/gcmd_classifier/datasets/extraction.py
19.5 kB
| """Deterministic, source-faithful PDF page, section, and block extraction.""" | |
| from __future__ import annotations | |
| import hashlib | |
| import io | |
| import re | |
| import time | |
| from collections.abc import Callable, Iterator | |
| from contextlib import AbstractContextManager | |
| from dataclasses import dataclass | |
| from typing import Any, Protocol | |
| import pdfplumber | |
| from gcmd_classifier.config import DatasetExtractionSettings | |
| from gcmd_classifier.datasets.documents import RetrievedREADME | |
| from gcmd_classifier.datasets.errors import DatasetExtractionError | |
| from gcmd_classifier.datasets.models import ( | |
| DatasetErrorRecord, | |
| DatasetExtractionManifest, | |
| DatasetWarning, | |
| ExtractedBlock, | |
| ExtractedPage, | |
| ExtractedSection, | |
| ExtractionLineage, | |
| ExtractionPreclassificationFailure, | |
| HeadingKind, | |
| PageExtractionStatus, | |
| SectionLabelSource, | |
| ) | |
| _NUMBERED_HEADING = re.compile(r"^(?:\d+(?:\.\d+)*[.)]?|[A-Z][.)])\s+\S") | |
| class PDFPage(Protocol): | |
| """Narrow passive page interface used by production and failure fakes.""" | |
| width: float | |
| height: float | |
| images: list[dict[str, Any]] | |
| lines: list[dict[str, Any]] | |
| rects: list[dict[str, Any]] | |
| def extract_text(self, **kwargs: Any) -> str | None: ... | |
| def extract_words(self, **kwargs: Any) -> list[dict[str, Any]]: ... | |
| class PDFDocument(Protocol): | |
| """Narrow passive document interface.""" | |
| pages: list[PDFPage] | |
| PDFOpener = Callable[[io.BytesIO], AbstractContextManager[PDFDocument]] | |
| class _Heading: | |
| page_number: int | |
| start: int | |
| end: int | |
| text: str | |
| kind: HeadingKind | |
| class _SectionDraft: | |
| section_id: str | |
| source_order: int | |
| heading: str | None | |
| normalized_heading: str | None | |
| heading_kind: HeadingKind | |
| label_source: SectionLabelSource | |
| heading_page_number: int | None | |
| heading_start: int | None | |
| heading_end: int | None | |
| page_start: int | |
| page_end: int | |
| block_ids: list[str] | |
| def _warning(code: str, message: str, **details: Any) -> DatasetWarning: | |
| return DatasetWarning( | |
| code=code, | |
| message=message, | |
| stage="extraction", | |
| details=details or None, | |
| ) | |
| def _fatal(code: str, message: str) -> DatasetExtractionError: | |
| return DatasetExtractionError(code, message) | |
| def _headings(text: str, page_number: int, policy: DatasetExtractionSettings) -> list[_Heading]: | |
| found: list[_Heading] = [] | |
| offset = 0 | |
| for line_with_ending in text.splitlines(keepends=True): | |
| line = line_with_ending.rstrip("\r\n") | |
| line_end = offset + len(line) | |
| stripped = line.strip() | |
| leading = len(line) - len(line.lstrip()) | |
| start = offset + leading | |
| kind: HeadingKind | None = None | |
| if 0 < len(stripped) <= policy.heading_max_characters: | |
| letters = [char for char in stripped if char.isalpha()] | |
| if _NUMBERED_HEADING.match(stripped): | |
| kind = HeadingKind.NUMBERED | |
| elif len(letters) >= 3 and stripped.upper() == stripped: | |
| kind = HeadingKind.ALL_CAPS | |
| elif stripped.endswith(":") and len(stripped.split()) <= 12: | |
| kind = HeadingKind.COLON_LABEL | |
| if kind is not None: | |
| found.append( | |
| _Heading( | |
| page_number=page_number, | |
| start=start, | |
| end=line_end - (len(line) - len(line.rstrip())), | |
| text=text[start:line_end].rstrip(), | |
| kind=kind, | |
| ) | |
| ) | |
| offset += len(line_with_ending) | |
| return found | |
| def _layout_warnings( | |
| page: PDFPage, text: str, page_number: int, policy: DatasetExtractionSettings | |
| ) -> list[DatasetWarning]: | |
| warnings: list[DatasetWarning] = [] | |
| if not text: | |
| if page.images: | |
| warnings.append( | |
| _warning( | |
| "IMAGE_ONLY_PAGE", | |
| "Page has images but no extractable text.", | |
| page_number=page_number, | |
| ) | |
| ) | |
| else: | |
| warnings.append( | |
| _warning( | |
| "BLANK_PAGE", "Page has no extractable text or images.", page_number=page_number | |
| ) | |
| ) | |
| return warnings | |
| if len(text) < policy.low_text_characters_per_page: | |
| warnings.append( | |
| _warning( | |
| "LOW_TEXT_COVERAGE", | |
| "Page has suspiciously little extracted text.", | |
| page_number=page_number, | |
| character_count=len(text), | |
| threshold=policy.low_text_characters_per_page, | |
| ) | |
| ) | |
| if len(page.lines) + len(page.rects) >= 8: | |
| warnings.append( | |
| _warning( | |
| "TABLE_LIKE_LAYOUT", | |
| "Page contains a table-like ruled layout.", | |
| page_number=page_number, | |
| ) | |
| ) | |
| try: | |
| words = page.extract_words( | |
| x_tolerance=policy.x_tolerance_points, y_tolerance=policy.y_tolerance_points | |
| ) | |
| except Exception: | |
| warnings.append( | |
| _warning( | |
| "LAYOUT_COORDINATES_UNAVAILABLE", | |
| "Word coordinates are unavailable for layout diagnostics.", | |
| page_number=page_number, | |
| ) | |
| ) | |
| return warnings | |
| centers = sorted( | |
| (float(word["x0"]) + float(word["x1"])) / 2 | |
| for word in words | |
| if "x0" in word and "x1" in word | |
| ) | |
| minimum = policy.multi_column_min_words_per_column | |
| if len(centers) >= minimum * 2: | |
| gaps = [ | |
| (centers[index + 1] - centers[index], index) | |
| for index in range(minimum - 1, len(centers) - minimum) | |
| ] | |
| if gaps and max(gaps)[0] >= float(page.width) * policy.multi_column_gap_ratio: | |
| warnings.extend( | |
| ( | |
| _warning( | |
| "LIKELY_MULTI_COLUMN", | |
| "Page layout likely contains multiple columns.", | |
| page_number=page_number, | |
| ), | |
| _warning( | |
| "SUSPICIOUS_READING_ORDER", | |
| "Extracted reading order may not match visual order.", | |
| page_number=page_number, | |
| ), | |
| ) | |
| ) | |
| return warnings | |
| def _split_ranges(start: int, end: int, text: str, maximum: int) -> Iterator[tuple[int, int]]: | |
| position = start | |
| while position < end: | |
| proposed = min(position + maximum, end) | |
| if proposed < end: | |
| newline = text.rfind("\n", position + 1, proposed + 1) | |
| if newline >= position: | |
| proposed = newline + 1 | |
| if proposed <= position: | |
| proposed = min(position + maximum, end) | |
| yield position, proposed | |
| position = proposed | |
| def _open_pdf(stream: io.BytesIO) -> AbstractContextManager[PDFDocument]: | |
| return pdfplumber.open(stream) | |
| def extract_pdf_manifest( | |
| source: RetrievedREADME, | |
| *, | |
| settings: DatasetExtractionSettings | None = None, | |
| pdf_opener: PDFOpener | None = None, | |
| monotonic: Callable[[], float] | None = None, | |
| ) -> DatasetExtractionManifest | ExtractionPreclassificationFailure: | |
| """Extract a validated Milestone 3 artifact or return a zero-inference fatal result.""" | |
| try: | |
| return _extract_pdf_manifest( | |
| source, | |
| settings=settings or DatasetExtractionSettings(), | |
| pdf_opener=pdf_opener or _open_pdf, | |
| monotonic=monotonic or time.monotonic, | |
| ) | |
| except DatasetExtractionError as exc: | |
| return ExtractionPreclassificationFailure( | |
| error=DatasetErrorRecord(code=exc.code, message=str(exc), stage="extraction") | |
| ) | |
| def _extract_pdf_manifest( | |
| source: RetrievedREADME, | |
| *, | |
| settings: DatasetExtractionSettings, | |
| pdf_opener: PDFOpener, | |
| monotonic: Callable[[], float], | |
| ) -> DatasetExtractionManifest: | |
| if not isinstance(source, RetrievedREADME): | |
| raise _fatal( | |
| "VALIDATED_PDF_REQUIRED", "Extraction requires a validated Milestone 3 PDF artifact." | |
| ) | |
| body = source.content | |
| record = source.record | |
| actual_hash = hashlib.sha256(body).hexdigest() | |
| if actual_hash != record.artifact.sha256 or len(body) != record.artifact.byte_size: | |
| raise _fatal("PDF_HASH_MISMATCH", "PDF bytes no longer match the validated artifact.") | |
| if record.detected_media_type != "application/pdf" or not record.document_validated: | |
| raise _fatal("VALIDATED_PDF_REQUIRED", "Artifact has not passed PDF validation.") | |
| started = monotonic() | |
| try: | |
| context = pdf_opener(io.BytesIO(body)) | |
| with context as pdf: | |
| if len(pdf.pages) > settings.max_pages: | |
| raise _fatal("PDF_PAGE_LIMIT_EXCEEDED", "PDF exceeds the configured page limit.") | |
| raw_pages: list[tuple[str, list[DatasetWarning], float | None, float | None]] = [] | |
| total_characters = 0 | |
| for page_number, page in enumerate(pdf.pages, start=1): | |
| if monotonic() - started > settings.timeout_seconds: | |
| raise _fatal( | |
| "PDF_EXTRACTION_TIMEOUT", "PDF extraction exceeded its time limit." | |
| ) | |
| try: | |
| text = page.extract_text( | |
| layout=False, | |
| x_tolerance=settings.x_tolerance_points, | |
| y_tolerance=settings.y_tolerance_points, | |
| ) | |
| except Exception as exc: | |
| raise _fatal( | |
| "PDF_PAGE_EXTRACTION_FAILED", f"PDF page {page_number} extraction failed." | |
| ) from exc | |
| if text is None: | |
| text = "" | |
| if not isinstance(text, str): | |
| raise _fatal( | |
| "PDF_PARTIAL_PAGE_FAILURE", | |
| f"PDF page {page_number} returned invalid partial text.", | |
| ) | |
| total_characters += len(text) | |
| if total_characters > settings.max_characters: | |
| raise _fatal( | |
| "PDF_CHARACTER_LIMIT_EXCEEDED", | |
| "Extracted text exceeds the configured character limit.", | |
| ) | |
| raw_pages.append( | |
| ( | |
| text, | |
| _layout_warnings(page, text, page_number, settings), | |
| float(page.width) if page.width else None, | |
| float(page.height) if page.height else None, | |
| ) | |
| ) | |
| except DatasetExtractionError: | |
| raise | |
| except Exception as exc: | |
| diagnostic_name = f"{type(exc).__name__}:{exc!r}".lower() | |
| if "password" in diagnostic_name or "encrypted" in diagnostic_name: | |
| raise _fatal("PDF_ENCRYPTED", "PDF is encrypted or password protected.") from exc | |
| raise _fatal("PDF_MALFORMED", "PDF is malformed or unreadable.") from exc | |
| if not raw_pages: | |
| raise _fatal("PDF_NO_PAGES", "PDF contains no pages.") | |
| if not any(text for text, _, _, _ in raw_pages): | |
| code = ( | |
| "PDF_IMAGE_ONLY" | |
| if any( | |
| any(w.code == "IMAGE_ONLY_PAGE" for w in warnings) | |
| for _, warnings, _, _ in raw_pages | |
| ) | |
| else "PDF_NO_EXTRACTABLE_TEXT" | |
| ) | |
| raise _fatal(code, "PDF contains no extractable text.") | |
| document_id = f"sha256:{actual_hash}" | |
| page_models: list[ExtractedPage] = [] | |
| headings_by_page: dict[int, list[_Heading]] = {} | |
| for page_number, (text, warnings, width, height) in enumerate(raw_pages, start=1): | |
| headings_by_page[page_number] = _headings(text, page_number, settings) | |
| status = ( | |
| PageExtractionStatus.EXTRACTED | |
| if text | |
| else ( | |
| PageExtractionStatus.IMAGE_ONLY | |
| if any(w.code == "IMAGE_ONLY_PAGE" for w in warnings) | |
| else PageExtractionStatus.BLANK | |
| ) | |
| ) | |
| page_models.append( | |
| ExtractedPage( | |
| page_id=f"{document_id}:p{page_number:04}", | |
| page_number=page_number, | |
| source_order=page_number - 1, | |
| text=text, | |
| text_sha256=hashlib.sha256(text.encode("utf-8")).hexdigest(), | |
| character_count=len(text), | |
| extraction_status=status, | |
| width_points=width, | |
| height_points=height, | |
| coordinate_system="pdf_points_top_origin" if width and height else None, | |
| warnings=tuple(warnings), | |
| ) | |
| ) | |
| sections: list[_SectionDraft] = [] | |
| blocks: list[ExtractedBlock] = [] | |
| active: _SectionDraft | None = None | |
| global_block_order = 0 | |
| page_block_ids: dict[int, list[str]] = {page.page_number: [] for page in page_models} | |
| for page in page_models: | |
| page_headings = headings_by_page[page.page_number] | |
| boundaries: list[tuple[int, _Heading | None]] = [] | |
| if not page_headings or page_headings[0].start > 0: | |
| boundaries.append((0, None)) | |
| boundaries.extend((heading.start, heading) for heading in page_headings) | |
| if not boundaries and not page.text: | |
| continue | |
| for boundary_index, (start, heading) in enumerate(boundaries): | |
| end = ( | |
| boundaries[boundary_index + 1][0] | |
| if boundary_index + 1 < len(boundaries) | |
| else len(page.text) | |
| ) | |
| if heading is not None: | |
| active = _SectionDraft( | |
| section_id=f"s{len(sections) + 1:04}", | |
| source_order=len(sections), | |
| heading=heading.text, | |
| normalized_heading=" ".join(heading.text.casefold().split()), | |
| heading_kind=heading.kind, | |
| label_source=SectionLabelSource.LITERAL, | |
| heading_page_number=page.page_number, | |
| heading_start=heading.start, | |
| heading_end=heading.end, | |
| page_start=page.page_number, | |
| page_end=page.page_number, | |
| block_ids=[], | |
| ) | |
| sections.append(active) | |
| elif active is None: | |
| active = _SectionDraft( | |
| section_id=f"s{len(sections) + 1:04}", | |
| source_order=len(sections), | |
| heading="Document body", | |
| normalized_heading="document body", | |
| heading_kind=HeadingKind.DERIVED_DOCUMENT, | |
| label_source=SectionLabelSource.DERIVED, | |
| heading_page_number=None, | |
| heading_start=None, | |
| heading_end=None, | |
| page_start=page.page_number, | |
| page_end=page.page_number, | |
| block_ids=[], | |
| ) | |
| sections.append(active) | |
| if active is None or start == end: | |
| continue | |
| active.page_end = page.page_number | |
| for block_start, block_end in _split_ranges( | |
| start, end, page.text, settings.max_block_characters | |
| ): | |
| if len(blocks) >= settings.max_blocks: | |
| raise _fatal( | |
| "PDF_BLOCK_LIMIT_EXCEEDED", "Extraction exceeds the configured block limit." | |
| ) | |
| block_id = f"p{page.page_number:04}-b{len(page_block_ids[page.page_number]) + 1:04}" | |
| block = ExtractedBlock( | |
| block_id=block_id, | |
| document_id=document_id, | |
| pdf_sha256=actual_hash, | |
| page_id=page.page_id, | |
| page_start=page.page_number, | |
| page_end=page.page_number, | |
| section_id=active.section_id, | |
| section_heading=active.heading, | |
| section_label_source=active.label_source, | |
| text=page.text[block_start:block_end], | |
| character_start=block_start, | |
| character_end=block_end, | |
| source_order=global_block_order, | |
| extraction_version=settings.extraction_version, | |
| splitting_policy_version=settings.segmentation_version, | |
| ) | |
| blocks.append(block) | |
| active.block_ids.append(block_id) | |
| page_block_ids[page.page_number].append(block_id) | |
| global_block_order += 1 | |
| if not blocks: | |
| raise _fatal("PDF_NO_EXTRACTABLE_TEXT", "PDF produced no extractable blocks.") | |
| pages = tuple( | |
| page.model_copy(update={"block_ids": tuple(page_block_ids[page.page_number])}) | |
| for page in page_models | |
| ) | |
| section_models = tuple( | |
| ExtractedSection( | |
| section_id=item.section_id, | |
| source_order=item.source_order, | |
| heading=item.heading, | |
| normalized_heading=item.normalized_heading, | |
| heading_kind=item.heading_kind, | |
| label_source=item.label_source, | |
| heading_page_number=item.heading_page_number, | |
| heading_character_start=item.heading_start, | |
| heading_character_end=item.heading_end, | |
| page_start=item.page_start, | |
| page_end=item.page_end, | |
| block_ids=tuple(item.block_ids), | |
| ) | |
| for item in sections | |
| ) | |
| lineage = ExtractionLineage( | |
| identity=record.identity, | |
| cmr_source_sha256=record.cmr_source_sha256, | |
| selected_candidate_id=record.selected_candidate_id, | |
| selected_source_index=record.selected_source_index, | |
| selected_source_entry=record.selected_source_entry, | |
| selected_readme_url=record.submitted_url, | |
| final_readme_url=record.final_url, | |
| retrieval_timestamp=record.retrieved_at, | |
| retrieval_status_code=record.status_code, | |
| public_address_validation=record.public_address_validation, | |
| document_validation_method=record.document_validation_method, | |
| ) | |
| all_warnings = tuple(warning for page in pages for warning in page.warnings) | |
| return DatasetExtractionManifest( | |
| schema_version="1.0", | |
| document=record.artifact, | |
| lineage=lineage, | |
| document_id=document_id, | |
| extraction_method="pdfplumber", | |
| extraction_version=settings.extraction_version, | |
| page_text_policy_version=settings.page_text_policy_version, | |
| heading_policy_version=settings.heading_policy_version, | |
| section_policy_version=settings.section_policy_version, | |
| segmentation_version=settings.segmentation_version, | |
| identifier_policy_version=settings.identifier_policy_version, | |
| separator_policy_version=settings.separator_policy_version, | |
| page_count=len(pages), | |
| character_count=sum(page.character_count for page in pages), | |
| block_count=len(blocks), | |
| configured_limits={ | |
| "max_pages": settings.max_pages, | |
| "max_characters": settings.max_characters, | |
| "max_blocks": settings.max_blocks, | |
| "max_block_characters": settings.max_block_characters, | |
| "timeout_seconds": int(settings.timeout_seconds), | |
| }, | |
| pages=pages, | |
| sections=section_models, | |
| blocks=tuple(blocks), | |
| warnings=all_warnings, | |
| ) | |