"""Deterministic, source-faithful PDF page, section, and block extraction.""" from __future__ import annotations import hashlib import io import re import time from collections.abc import Callable, Iterator from contextlib import AbstractContextManager from dataclasses import dataclass from typing import Any, Protocol import pdfplumber from gcmd_classifier.config import DatasetExtractionSettings from gcmd_classifier.datasets.documents import RetrievedREADME from gcmd_classifier.datasets.errors import DatasetExtractionError from gcmd_classifier.datasets.models import ( DatasetErrorRecord, DatasetExtractionManifest, DatasetWarning, ExtractedBlock, ExtractedPage, ExtractedSection, ExtractionLineage, ExtractionPreclassificationFailure, HeadingKind, PageExtractionStatus, SectionLabelSource, ) _NUMBERED_HEADING = re.compile(r"^(?:\d+(?:\.\d+)*[.)]?|[A-Z][.)])\s+\S") class PDFPage(Protocol): """Narrow passive page interface used by production and failure fakes.""" width: float height: float images: list[dict[str, Any]] lines: list[dict[str, Any]] rects: list[dict[str, Any]] def extract_text(self, **kwargs: Any) -> str | None: ... def extract_words(self, **kwargs: Any) -> list[dict[str, Any]]: ... class PDFDocument(Protocol): """Narrow passive document interface.""" pages: list[PDFPage] PDFOpener = Callable[[io.BytesIO], AbstractContextManager[PDFDocument]] @dataclass(frozen=True) class _Heading: page_number: int start: int end: int text: str kind: HeadingKind @dataclass class _SectionDraft: section_id: str source_order: int heading: str | None normalized_heading: str | None heading_kind: HeadingKind label_source: SectionLabelSource heading_page_number: int | None heading_start: int | None heading_end: int | None page_start: int page_end: int block_ids: list[str] def _warning(code: str, message: str, **details: Any) -> DatasetWarning: return DatasetWarning( code=code, message=message, stage="extraction", details=details or None, ) def _fatal(code: str, message: str) -> DatasetExtractionError: return DatasetExtractionError(code, message) def _headings(text: str, page_number: int, policy: DatasetExtractionSettings) -> list[_Heading]: found: list[_Heading] = [] offset = 0 for line_with_ending in text.splitlines(keepends=True): line = line_with_ending.rstrip("\r\n") line_end = offset + len(line) stripped = line.strip() leading = len(line) - len(line.lstrip()) start = offset + leading kind: HeadingKind | None = None if 0 < len(stripped) <= policy.heading_max_characters: letters = [char for char in stripped if char.isalpha()] if _NUMBERED_HEADING.match(stripped): kind = HeadingKind.NUMBERED elif len(letters) >= 3 and stripped.upper() == stripped: kind = HeadingKind.ALL_CAPS elif stripped.endswith(":") and len(stripped.split()) <= 12: kind = HeadingKind.COLON_LABEL if kind is not None: found.append( _Heading( page_number=page_number, start=start, end=line_end - (len(line) - len(line.rstrip())), text=text[start:line_end].rstrip(), kind=kind, ) ) offset += len(line_with_ending) return found def _layout_warnings( page: PDFPage, text: str, page_number: int, policy: DatasetExtractionSettings ) -> list[DatasetWarning]: warnings: list[DatasetWarning] = [] if not text: if page.images: warnings.append( _warning( "IMAGE_ONLY_PAGE", "Page has images but no extractable text.", page_number=page_number, ) ) else: warnings.append( _warning( "BLANK_PAGE", "Page has no extractable text or images.", page_number=page_number ) ) return warnings if len(text) < policy.low_text_characters_per_page: warnings.append( _warning( "LOW_TEXT_COVERAGE", "Page has suspiciously little extracted text.", page_number=page_number, character_count=len(text), threshold=policy.low_text_characters_per_page, ) ) if len(page.lines) + len(page.rects) >= 8: warnings.append( _warning( "TABLE_LIKE_LAYOUT", "Page contains a table-like ruled layout.", page_number=page_number, ) ) try: words = page.extract_words( x_tolerance=policy.x_tolerance_points, y_tolerance=policy.y_tolerance_points ) except Exception: warnings.append( _warning( "LAYOUT_COORDINATES_UNAVAILABLE", "Word coordinates are unavailable for layout diagnostics.", page_number=page_number, ) ) return warnings centers = sorted( (float(word["x0"]) + float(word["x1"])) / 2 for word in words if "x0" in word and "x1" in word ) minimum = policy.multi_column_min_words_per_column if len(centers) >= minimum * 2: gaps = [ (centers[index + 1] - centers[index], index) for index in range(minimum - 1, len(centers) - minimum) ] if gaps and max(gaps)[0] >= float(page.width) * policy.multi_column_gap_ratio: warnings.extend( ( _warning( "LIKELY_MULTI_COLUMN", "Page layout likely contains multiple columns.", page_number=page_number, ), _warning( "SUSPICIOUS_READING_ORDER", "Extracted reading order may not match visual order.", page_number=page_number, ), ) ) return warnings def _split_ranges(start: int, end: int, text: str, maximum: int) -> Iterator[tuple[int, int]]: position = start while position < end: proposed = min(position + maximum, end) if proposed < end: newline = text.rfind("\n", position + 1, proposed + 1) if newline >= position: proposed = newline + 1 if proposed <= position: proposed = min(position + maximum, end) yield position, proposed position = proposed def _open_pdf(stream: io.BytesIO) -> AbstractContextManager[PDFDocument]: return pdfplumber.open(stream) def extract_pdf_manifest( source: RetrievedREADME, *, settings: DatasetExtractionSettings | None = None, pdf_opener: PDFOpener | None = None, monotonic: Callable[[], float] | None = None, ) -> DatasetExtractionManifest | ExtractionPreclassificationFailure: """Extract a validated Milestone 3 artifact or return a zero-inference fatal result.""" try: return _extract_pdf_manifest( source, settings=settings or DatasetExtractionSettings(), pdf_opener=pdf_opener or _open_pdf, monotonic=monotonic or time.monotonic, ) except DatasetExtractionError as exc: return ExtractionPreclassificationFailure( error=DatasetErrorRecord(code=exc.code, message=str(exc), stage="extraction") ) def _extract_pdf_manifest( source: RetrievedREADME, *, settings: DatasetExtractionSettings, pdf_opener: PDFOpener, monotonic: Callable[[], float], ) -> DatasetExtractionManifest: if not isinstance(source, RetrievedREADME): raise _fatal( "VALIDATED_PDF_REQUIRED", "Extraction requires a validated Milestone 3 PDF artifact." ) body = source.content record = source.record actual_hash = hashlib.sha256(body).hexdigest() if actual_hash != record.artifact.sha256 or len(body) != record.artifact.byte_size: raise _fatal("PDF_HASH_MISMATCH", "PDF bytes no longer match the validated artifact.") if record.detected_media_type != "application/pdf" or not record.document_validated: raise _fatal("VALIDATED_PDF_REQUIRED", "Artifact has not passed PDF validation.") started = monotonic() try: context = pdf_opener(io.BytesIO(body)) with context as pdf: if len(pdf.pages) > settings.max_pages: raise _fatal("PDF_PAGE_LIMIT_EXCEEDED", "PDF exceeds the configured page limit.") raw_pages: list[tuple[str, list[DatasetWarning], float | None, float | None]] = [] total_characters = 0 for page_number, page in enumerate(pdf.pages, start=1): if monotonic() - started > settings.timeout_seconds: raise _fatal( "PDF_EXTRACTION_TIMEOUT", "PDF extraction exceeded its time limit." ) try: text = page.extract_text( layout=False, x_tolerance=settings.x_tolerance_points, y_tolerance=settings.y_tolerance_points, ) except Exception as exc: raise _fatal( "PDF_PAGE_EXTRACTION_FAILED", f"PDF page {page_number} extraction failed." ) from exc if text is None: text = "" if not isinstance(text, str): raise _fatal( "PDF_PARTIAL_PAGE_FAILURE", f"PDF page {page_number} returned invalid partial text.", ) total_characters += len(text) if total_characters > settings.max_characters: raise _fatal( "PDF_CHARACTER_LIMIT_EXCEEDED", "Extracted text exceeds the configured character limit.", ) raw_pages.append( ( text, _layout_warnings(page, text, page_number, settings), float(page.width) if page.width else None, float(page.height) if page.height else None, ) ) except DatasetExtractionError: raise except Exception as exc: diagnostic_name = f"{type(exc).__name__}:{exc!r}".lower() if "password" in diagnostic_name or "encrypted" in diagnostic_name: raise _fatal("PDF_ENCRYPTED", "PDF is encrypted or password protected.") from exc raise _fatal("PDF_MALFORMED", "PDF is malformed or unreadable.") from exc if not raw_pages: raise _fatal("PDF_NO_PAGES", "PDF contains no pages.") if not any(text for text, _, _, _ in raw_pages): code = ( "PDF_IMAGE_ONLY" if any( any(w.code == "IMAGE_ONLY_PAGE" for w in warnings) for _, warnings, _, _ in raw_pages ) else "PDF_NO_EXTRACTABLE_TEXT" ) raise _fatal(code, "PDF contains no extractable text.") document_id = f"sha256:{actual_hash}" page_models: list[ExtractedPage] = [] headings_by_page: dict[int, list[_Heading]] = {} for page_number, (text, warnings, width, height) in enumerate(raw_pages, start=1): headings_by_page[page_number] = _headings(text, page_number, settings) status = ( PageExtractionStatus.EXTRACTED if text else ( PageExtractionStatus.IMAGE_ONLY if any(w.code == "IMAGE_ONLY_PAGE" for w in warnings) else PageExtractionStatus.BLANK ) ) page_models.append( ExtractedPage( page_id=f"{document_id}:p{page_number:04}", page_number=page_number, source_order=page_number - 1, text=text, text_sha256=hashlib.sha256(text.encode("utf-8")).hexdigest(), character_count=len(text), extraction_status=status, width_points=width, height_points=height, coordinate_system="pdf_points_top_origin" if width and height else None, warnings=tuple(warnings), ) ) sections: list[_SectionDraft] = [] blocks: list[ExtractedBlock] = [] active: _SectionDraft | None = None global_block_order = 0 page_block_ids: dict[int, list[str]] = {page.page_number: [] for page in page_models} for page in page_models: page_headings = headings_by_page[page.page_number] boundaries: list[tuple[int, _Heading | None]] = [] if not page_headings or page_headings[0].start > 0: boundaries.append((0, None)) boundaries.extend((heading.start, heading) for heading in page_headings) if not boundaries and not page.text: continue for boundary_index, (start, heading) in enumerate(boundaries): end = ( boundaries[boundary_index + 1][0] if boundary_index + 1 < len(boundaries) else len(page.text) ) if heading is not None: active = _SectionDraft( section_id=f"s{len(sections) + 1:04}", source_order=len(sections), heading=heading.text, normalized_heading=" ".join(heading.text.casefold().split()), heading_kind=heading.kind, label_source=SectionLabelSource.LITERAL, heading_page_number=page.page_number, heading_start=heading.start, heading_end=heading.end, page_start=page.page_number, page_end=page.page_number, block_ids=[], ) sections.append(active) elif active is None: active = _SectionDraft( section_id=f"s{len(sections) + 1:04}", source_order=len(sections), heading="Document body", normalized_heading="document body", heading_kind=HeadingKind.DERIVED_DOCUMENT, label_source=SectionLabelSource.DERIVED, heading_page_number=None, heading_start=None, heading_end=None, page_start=page.page_number, page_end=page.page_number, block_ids=[], ) sections.append(active) if active is None or start == end: continue active.page_end = page.page_number for block_start, block_end in _split_ranges( start, end, page.text, settings.max_block_characters ): if len(blocks) >= settings.max_blocks: raise _fatal( "PDF_BLOCK_LIMIT_EXCEEDED", "Extraction exceeds the configured block limit." ) block_id = f"p{page.page_number:04}-b{len(page_block_ids[page.page_number]) + 1:04}" block = ExtractedBlock( block_id=block_id, document_id=document_id, pdf_sha256=actual_hash, page_id=page.page_id, page_start=page.page_number, page_end=page.page_number, section_id=active.section_id, section_heading=active.heading, section_label_source=active.label_source, text=page.text[block_start:block_end], character_start=block_start, character_end=block_end, source_order=global_block_order, extraction_version=settings.extraction_version, splitting_policy_version=settings.segmentation_version, ) blocks.append(block) active.block_ids.append(block_id) page_block_ids[page.page_number].append(block_id) global_block_order += 1 if not blocks: raise _fatal("PDF_NO_EXTRACTABLE_TEXT", "PDF produced no extractable blocks.") pages = tuple( page.model_copy(update={"block_ids": tuple(page_block_ids[page.page_number])}) for page in page_models ) section_models = tuple( ExtractedSection( section_id=item.section_id, source_order=item.source_order, heading=item.heading, normalized_heading=item.normalized_heading, heading_kind=item.heading_kind, label_source=item.label_source, heading_page_number=item.heading_page_number, heading_character_start=item.heading_start, heading_character_end=item.heading_end, page_start=item.page_start, page_end=item.page_end, block_ids=tuple(item.block_ids), ) for item in sections ) lineage = ExtractionLineage( identity=record.identity, cmr_source_sha256=record.cmr_source_sha256, selected_candidate_id=record.selected_candidate_id, selected_source_index=record.selected_source_index, selected_source_entry=record.selected_source_entry, selected_readme_url=record.submitted_url, final_readme_url=record.final_url, retrieval_timestamp=record.retrieved_at, retrieval_status_code=record.status_code, public_address_validation=record.public_address_validation, document_validation_method=record.document_validation_method, ) all_warnings = tuple(warning for page in pages for warning in page.warnings) return DatasetExtractionManifest( schema_version="1.0", document=record.artifact, lineage=lineage, document_id=document_id, extraction_method="pdfplumber", extraction_version=settings.extraction_version, page_text_policy_version=settings.page_text_policy_version, heading_policy_version=settings.heading_policy_version, section_policy_version=settings.section_policy_version, segmentation_version=settings.segmentation_version, identifier_policy_version=settings.identifier_policy_version, separator_policy_version=settings.separator_policy_version, page_count=len(pages), character_count=sum(page.character_count for page in pages), block_count=len(blocks), configured_limits={ "max_pages": settings.max_pages, "max_characters": settings.max_characters, "max_blocks": settings.max_blocks, "max_block_characters": settings.max_block_characters, "timeout_seconds": int(settings.timeout_seconds), }, pages=pages, sections=section_models, blocks=tuple(blocks), warnings=all_warnings, )