igerasimov's picture
Deploy Phase 2 dataset classifier (part 19)
58c2da3 verified
Raw History Blame Contribute Delete
19.5 kB
"""Deterministic, source-faithful PDF page, section, and block extraction."""
from __future__ import annotations
import hashlib
import io
import re
import time
from collections.abc import Callable, Iterator
from contextlib import AbstractContextManager
from dataclasses import dataclass
from typing import Any, Protocol
import pdfplumber
from gcmd_classifier.config import DatasetExtractionSettings
from gcmd_classifier.datasets.documents import RetrievedREADME
from gcmd_classifier.datasets.errors import DatasetExtractionError
from gcmd_classifier.datasets.models import (
DatasetErrorRecord,
DatasetExtractionManifest,
DatasetWarning,
ExtractedBlock,
ExtractedPage,
ExtractedSection,
ExtractionLineage,
ExtractionPreclassificationFailure,
HeadingKind,
PageExtractionStatus,
SectionLabelSource,
)
_NUMBERED_HEADING = re.compile(r"^(?:\d+(?:\.\d+)*[.)]?|[A-Z][.)])\s+\S")
class PDFPage(Protocol):
"""Narrow passive page interface used by production and failure fakes."""
width: float
height: float
images: list[dict[str, Any]]
lines: list[dict[str, Any]]
rects: list[dict[str, Any]]
def extract_text(self, **kwargs: Any) -> str | None: ...
def extract_words(self, **kwargs: Any) -> list[dict[str, Any]]: ...
class PDFDocument(Protocol):
"""Narrow passive document interface."""
pages: list[PDFPage]
PDFOpener = Callable[[io.BytesIO], AbstractContextManager[PDFDocument]]
@dataclass(frozen=True)
class _Heading:
page_number: int
start: int
end: int
text: str
kind: HeadingKind
@dataclass
class _SectionDraft:
section_id: str
source_order: int
heading: str | None
normalized_heading: str | None
heading_kind: HeadingKind
label_source: SectionLabelSource
heading_page_number: int | None
heading_start: int | None
heading_end: int | None
page_start: int
page_end: int
block_ids: list[str]
def _warning(code: str, message: str, **details: Any) -> DatasetWarning:
return DatasetWarning(
code=code,
message=message,
stage="extraction",
details=details or None,
)
def _fatal(code: str, message: str) -> DatasetExtractionError:
return DatasetExtractionError(code, message)
def _headings(text: str, page_number: int, policy: DatasetExtractionSettings) -> list[_Heading]:
found: list[_Heading] = []
offset = 0
for line_with_ending in text.splitlines(keepends=True):
line = line_with_ending.rstrip("\r\n")
line_end = offset + len(line)
stripped = line.strip()
leading = len(line) - len(line.lstrip())
start = offset + leading
kind: HeadingKind | None = None
if 0 < len(stripped) <= policy.heading_max_characters:
letters = [char for char in stripped if char.isalpha()]
if _NUMBERED_HEADING.match(stripped):
kind = HeadingKind.NUMBERED
elif len(letters) >= 3 and stripped.upper() == stripped:
kind = HeadingKind.ALL_CAPS
elif stripped.endswith(":") and len(stripped.split()) <= 12:
kind = HeadingKind.COLON_LABEL
if kind is not None:
found.append(
_Heading(
page_number=page_number,
start=start,
end=line_end - (len(line) - len(line.rstrip())),
text=text[start:line_end].rstrip(),
kind=kind,
)
)
offset += len(line_with_ending)
return found
def _layout_warnings(
page: PDFPage, text: str, page_number: int, policy: DatasetExtractionSettings
) -> list[DatasetWarning]:
warnings: list[DatasetWarning] = []
if not text:
if page.images:
warnings.append(
_warning(
"IMAGE_ONLY_PAGE",
"Page has images but no extractable text.",
page_number=page_number,
)
)
else:
warnings.append(
_warning(
"BLANK_PAGE", "Page has no extractable text or images.", page_number=page_number
)
)
return warnings
if len(text) < policy.low_text_characters_per_page:
warnings.append(
_warning(
"LOW_TEXT_COVERAGE",
"Page has suspiciously little extracted text.",
page_number=page_number,
character_count=len(text),
threshold=policy.low_text_characters_per_page,
)
)
if len(page.lines) + len(page.rects) >= 8:
warnings.append(
_warning(
"TABLE_LIKE_LAYOUT",
"Page contains a table-like ruled layout.",
page_number=page_number,
)
)
try:
words = page.extract_words(
x_tolerance=policy.x_tolerance_points, y_tolerance=policy.y_tolerance_points
)
except Exception:
warnings.append(
_warning(
"LAYOUT_COORDINATES_UNAVAILABLE",
"Word coordinates are unavailable for layout diagnostics.",
page_number=page_number,
)
)
return warnings
centers = sorted(
(float(word["x0"]) + float(word["x1"])) / 2
for word in words
if "x0" in word and "x1" in word
)
minimum = policy.multi_column_min_words_per_column
if len(centers) >= minimum * 2:
gaps = [
(centers[index + 1] - centers[index], index)
for index in range(minimum - 1, len(centers) - minimum)
]
if gaps and max(gaps)[0] >= float(page.width) * policy.multi_column_gap_ratio:
warnings.extend(
(
_warning(
"LIKELY_MULTI_COLUMN",
"Page layout likely contains multiple columns.",
page_number=page_number,
),
_warning(
"SUSPICIOUS_READING_ORDER",
"Extracted reading order may not match visual order.",
page_number=page_number,
),
)
)
return warnings
def _split_ranges(start: int, end: int, text: str, maximum: int) -> Iterator[tuple[int, int]]:
position = start
while position < end:
proposed = min(position + maximum, end)
if proposed < end:
newline = text.rfind("\n", position + 1, proposed + 1)
if newline >= position:
proposed = newline + 1
if proposed <= position:
proposed = min(position + maximum, end)
yield position, proposed
position = proposed
def _open_pdf(stream: io.BytesIO) -> AbstractContextManager[PDFDocument]:
return pdfplumber.open(stream)
def extract_pdf_manifest(
source: RetrievedREADME,
*,
settings: DatasetExtractionSettings | None = None,
pdf_opener: PDFOpener | None = None,
monotonic: Callable[[], float] | None = None,
) -> DatasetExtractionManifest | ExtractionPreclassificationFailure:
"""Extract a validated Milestone 3 artifact or return a zero-inference fatal result."""
try:
return _extract_pdf_manifest(
source,
settings=settings or DatasetExtractionSettings(),
pdf_opener=pdf_opener or _open_pdf,
monotonic=monotonic or time.monotonic,
)
except DatasetExtractionError as exc:
return ExtractionPreclassificationFailure(
error=DatasetErrorRecord(code=exc.code, message=str(exc), stage="extraction")
)
def _extract_pdf_manifest(
source: RetrievedREADME,
*,
settings: DatasetExtractionSettings,
pdf_opener: PDFOpener,
monotonic: Callable[[], float],
) -> DatasetExtractionManifest:
if not isinstance(source, RetrievedREADME):
raise _fatal(
"VALIDATED_PDF_REQUIRED", "Extraction requires a validated Milestone 3 PDF artifact."
)
body = source.content
record = source.record
actual_hash = hashlib.sha256(body).hexdigest()
if actual_hash != record.artifact.sha256 or len(body) != record.artifact.byte_size:
raise _fatal("PDF_HASH_MISMATCH", "PDF bytes no longer match the validated artifact.")
if record.detected_media_type != "application/pdf" or not record.document_validated:
raise _fatal("VALIDATED_PDF_REQUIRED", "Artifact has not passed PDF validation.")
started = monotonic()
try:
context = pdf_opener(io.BytesIO(body))
with context as pdf:
if len(pdf.pages) > settings.max_pages:
raise _fatal("PDF_PAGE_LIMIT_EXCEEDED", "PDF exceeds the configured page limit.")
raw_pages: list[tuple[str, list[DatasetWarning], float | None, float | None]] = []
total_characters = 0
for page_number, page in enumerate(pdf.pages, start=1):
if monotonic() - started > settings.timeout_seconds:
raise _fatal(
"PDF_EXTRACTION_TIMEOUT", "PDF extraction exceeded its time limit."
)
try:
text = page.extract_text(
layout=False,
x_tolerance=settings.x_tolerance_points,
y_tolerance=settings.y_tolerance_points,
)
except Exception as exc:
raise _fatal(
"PDF_PAGE_EXTRACTION_FAILED", f"PDF page {page_number} extraction failed."
) from exc
if text is None:
text = ""
if not isinstance(text, str):
raise _fatal(
"PDF_PARTIAL_PAGE_FAILURE",
f"PDF page {page_number} returned invalid partial text.",
)
total_characters += len(text)
if total_characters > settings.max_characters:
raise _fatal(
"PDF_CHARACTER_LIMIT_EXCEEDED",
"Extracted text exceeds the configured character limit.",
)
raw_pages.append(
(
text,
_layout_warnings(page, text, page_number, settings),
float(page.width) if page.width else None,
float(page.height) if page.height else None,
)
)
except DatasetExtractionError:
raise
except Exception as exc:
diagnostic_name = f"{type(exc).__name__}:{exc!r}".lower()
if "password" in diagnostic_name or "encrypted" in diagnostic_name:
raise _fatal("PDF_ENCRYPTED", "PDF is encrypted or password protected.") from exc
raise _fatal("PDF_MALFORMED", "PDF is malformed or unreadable.") from exc
if not raw_pages:
raise _fatal("PDF_NO_PAGES", "PDF contains no pages.")
if not any(text for text, _, _, _ in raw_pages):
code = (
"PDF_IMAGE_ONLY"
if any(
any(w.code == "IMAGE_ONLY_PAGE" for w in warnings)
for _, warnings, _, _ in raw_pages
)
else "PDF_NO_EXTRACTABLE_TEXT"
)
raise _fatal(code, "PDF contains no extractable text.")
document_id = f"sha256:{actual_hash}"
page_models: list[ExtractedPage] = []
headings_by_page: dict[int, list[_Heading]] = {}
for page_number, (text, warnings, width, height) in enumerate(raw_pages, start=1):
headings_by_page[page_number] = _headings(text, page_number, settings)
status = (
PageExtractionStatus.EXTRACTED
if text
else (
PageExtractionStatus.IMAGE_ONLY
if any(w.code == "IMAGE_ONLY_PAGE" for w in warnings)
else PageExtractionStatus.BLANK
)
)
page_models.append(
ExtractedPage(
page_id=f"{document_id}:p{page_number:04}",
page_number=page_number,
source_order=page_number - 1,
text=text,
text_sha256=hashlib.sha256(text.encode("utf-8")).hexdigest(),
character_count=len(text),
extraction_status=status,
width_points=width,
height_points=height,
coordinate_system="pdf_points_top_origin" if width and height else None,
warnings=tuple(warnings),
)
)
sections: list[_SectionDraft] = []
blocks: list[ExtractedBlock] = []
active: _SectionDraft | None = None
global_block_order = 0
page_block_ids: dict[int, list[str]] = {page.page_number: [] for page in page_models}
for page in page_models:
page_headings = headings_by_page[page.page_number]
boundaries: list[tuple[int, _Heading | None]] = []
if not page_headings or page_headings[0].start > 0:
boundaries.append((0, None))
boundaries.extend((heading.start, heading) for heading in page_headings)
if not boundaries and not page.text:
continue
for boundary_index, (start, heading) in enumerate(boundaries):
end = (
boundaries[boundary_index + 1][0]
if boundary_index + 1 < len(boundaries)
else len(page.text)
)
if heading is not None:
active = _SectionDraft(
section_id=f"s{len(sections) + 1:04}",
source_order=len(sections),
heading=heading.text,
normalized_heading=" ".join(heading.text.casefold().split()),
heading_kind=heading.kind,
label_source=SectionLabelSource.LITERAL,
heading_page_number=page.page_number,
heading_start=heading.start,
heading_end=heading.end,
page_start=page.page_number,
page_end=page.page_number,
block_ids=[],
)
sections.append(active)
elif active is None:
active = _SectionDraft(
section_id=f"s{len(sections) + 1:04}",
source_order=len(sections),
heading="Document body",
normalized_heading="document body",
heading_kind=HeadingKind.DERIVED_DOCUMENT,
label_source=SectionLabelSource.DERIVED,
heading_page_number=None,
heading_start=None,
heading_end=None,
page_start=page.page_number,
page_end=page.page_number,
block_ids=[],
)
sections.append(active)
if active is None or start == end:
continue
active.page_end = page.page_number
for block_start, block_end in _split_ranges(
start, end, page.text, settings.max_block_characters
):
if len(blocks) >= settings.max_blocks:
raise _fatal(
"PDF_BLOCK_LIMIT_EXCEEDED", "Extraction exceeds the configured block limit."
)
block_id = f"p{page.page_number:04}-b{len(page_block_ids[page.page_number]) + 1:04}"
block = ExtractedBlock(
block_id=block_id,
document_id=document_id,
pdf_sha256=actual_hash,
page_id=page.page_id,
page_start=page.page_number,
page_end=page.page_number,
section_id=active.section_id,
section_heading=active.heading,
section_label_source=active.label_source,
text=page.text[block_start:block_end],
character_start=block_start,
character_end=block_end,
source_order=global_block_order,
extraction_version=settings.extraction_version,
splitting_policy_version=settings.segmentation_version,
)
blocks.append(block)
active.block_ids.append(block_id)
page_block_ids[page.page_number].append(block_id)
global_block_order += 1
if not blocks:
raise _fatal("PDF_NO_EXTRACTABLE_TEXT", "PDF produced no extractable blocks.")
pages = tuple(
page.model_copy(update={"block_ids": tuple(page_block_ids[page.page_number])})
for page in page_models
)
section_models = tuple(
ExtractedSection(
section_id=item.section_id,
source_order=item.source_order,
heading=item.heading,
normalized_heading=item.normalized_heading,
heading_kind=item.heading_kind,
label_source=item.label_source,
heading_page_number=item.heading_page_number,
heading_character_start=item.heading_start,
heading_character_end=item.heading_end,
page_start=item.page_start,
page_end=item.page_end,
block_ids=tuple(item.block_ids),
)
for item in sections
)
lineage = ExtractionLineage(
identity=record.identity,
cmr_source_sha256=record.cmr_source_sha256,
selected_candidate_id=record.selected_candidate_id,
selected_source_index=record.selected_source_index,
selected_source_entry=record.selected_source_entry,
selected_readme_url=record.submitted_url,
final_readme_url=record.final_url,
retrieval_timestamp=record.retrieved_at,
retrieval_status_code=record.status_code,
public_address_validation=record.public_address_validation,
document_validation_method=record.document_validation_method,
)
all_warnings = tuple(warning for page in pages for warning in page.warnings)
return DatasetExtractionManifest(
schema_version="1.0",
document=record.artifact,
lineage=lineage,
document_id=document_id,
extraction_method="pdfplumber",
extraction_version=settings.extraction_version,
page_text_policy_version=settings.page_text_policy_version,
heading_policy_version=settings.heading_policy_version,
section_policy_version=settings.section_policy_version,
segmentation_version=settings.segmentation_version,
identifier_policy_version=settings.identifier_policy_version,
separator_policy_version=settings.separator_policy_version,
page_count=len(pages),
character_count=sum(page.character_count for page in pages),
block_count=len(blocks),
configured_limits={
"max_pages": settings.max_pages,
"max_characters": settings.max_characters,
"max_blocks": settings.max_blocks,
"max_block_characters": settings.max_block_characters,
"timeout_seconds": int(settings.timeout_seconds),
},
pages=pages,
sections=section_models,
blocks=tuple(blocks),
warnings=all_warnings,
)