Download src/knowledge_extraction/adapter.py from DataEyond/Agentic-Service-Data-Eyond-Catalog: direct link, hf CLI and curl.
- Browser
- Download file 11.3 kB
-
https://huggingface.co/spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/resolve/main/src/knowledge_extraction/adapter.py
- Command line
-
hf download hf://spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/src/knowledge_extraction/adapter.py
-
curl -L -o adapter.py https://huggingface.co/spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/resolve/main/src/knowledge_extraction/adapter.py
11.3 kB
| """Seam adapter: parsed-document artifact → the pipeline's internal `Chunk`. | |
| **This is the only file that knows the seam's shape.** Every stage downstream | |
| depends on `models.Chunk` alone, so when the artifact contract settles with | |
| Sofhia the change lands here and nowhere else. | |
| The seam is still under discussion (KNOWLEDGE_PIPELINE_TODO.md §3), so this | |
| reads defensively: it accepts either the draft's field names or the prototype's, | |
| takes plain dicts, and tolerates missing optional structure. It deliberately | |
| does **not** accept a file path — extraction never opens a document. That | |
| constraint is the point of the split, not an implementation detail. | |
| Two things it must never do: | |
| - reflow, strip or whitespace-normalise `text`. Span validation locates quoted | |
| spans literally inside it; cleaning the text makes the lookup fail and the | |
| field go silently null, which presents as a bad model. | |
| - infer a page number it was not given. A wrong page sends the reviewer to the | |
| wrong part of the document, which is worse than no page at all. | |
| """ | |
| from __future__ import annotations | |
| import hashlib | |
| import json | |
| from typing import Any | |
| from src.middlewares.logging import get_logger | |
| from .models import Chunk, ChunkAsset, ParsedDoc | |
| logger = get_logger("knowledge_adapter") | |
| # Field names accepted for the same concept. `page_idx`/`page_idxs` are the | |
| # parsing half's contract; the rest are earlier drafts and the prototype's shape, | |
| # kept so old fixtures still load. | |
| # | |
| # Page numbers are 0-BASED throughout, exactly as the parser reports them. No | |
| # conversion happens anywhere in this pipeline: converting to 1-based is the | |
| # review UI's job, done once at display time. An off-by-one here would be | |
| # invisible until an expert opened the wrong page. | |
| _PAGE_KEYS = ("page_idx", "page_start", "page") | |
| _PAGES_KEYS = ("page_idxs", "pages", "page_list") | |
| _PAGE_END_KEYS = ("page_idx_end", "page_end") | |
| # ── Seam version gate (N6b) ────────────────────────────────────────────── | |
| # | |
| # The artifact's SHAPE version, as `major.minor`. Parsing owns the contract and | |
| # bumps `SCHEMA_VERSION` in `knowledge_parsing/contracts.py`; this is the range | |
| # extraction knows how to read. | |
| # | |
| # Deliberately a literal rather than an import: extraction must never depend on | |
| # the parsing package at runtime — that independence is the whole point of the | |
| # seam. Drift is caught instead by a local seam test that imports | |
| # `knowledge_parsing.contracts` directly and asserts the two agree, so a bump on | |
| # either side fails loudly in tests rather than silently at read time. | |
| SUPPORTED_SCHEMA = (0, 4) | |
| class IncompatibleArtifactError(ValueError): | |
| """The artifact declares a shape this code cannot read. | |
| Raised rather than degraded on purpose. Every other seam in this repo | |
| tolerates missing structure, because a chunk without a heading is still | |
| usable — but a shape we do not understand is not a smaller version of the | |
| right input, it is a different input. Reading it anyway produces entries | |
| that look fine and are wrong, which is the single most expensive failure | |
| this pipeline can have. | |
| """ | |
| def _parse_version(declared: str) -> tuple[int, int] | None: | |
| parts = str(declared).split(".") | |
| try: | |
| return int(parts[0]), int(parts[1]) | |
| except (IndexError, ValueError): | |
| return None | |
| def check_schema_version(declared: Any) -> None: | |
| """Gate one artifact's declared `schema_version`. Raises or returns None. | |
| - **Major mismatch → reject.** A major bump is by definition breaking. | |
| - **Newer minor → reject.** The artifact carries structure this code has | |
| never seen; guessing at it is how a silent misread starts. | |
| - **Older minor, or any patch → accept.** Minor bumps have been additive by | |
| convention (0.3.0 → 0.3.3 added `source_title`, `normaliser_version` and | |
| `page_no`), and the committed eval fixture is deliberately an older minor | |
| than the current parser, so rejecting those would break the baseline | |
| measurement for no safety gain. | |
| - **Absent → accept with a warning.** Bare chunk lists and pre-envelope | |
| fixtures carry no version; the CLI already refuses unlabelled artifacts | |
| where labelling matters. | |
| """ | |
| if declared in (None, ""): | |
| logger.warning( | |
| "artifact_schema_version_missing", | |
| extra={"supported": ".".join(map(str, SUPPORTED_SCHEMA))}, | |
| ) | |
| return | |
| parsed = _parse_version(declared) | |
| if parsed is None: | |
| raise IncompatibleArtifactError( | |
| f"artifact schema_version {declared!r} is not a major.minor version" | |
| ) | |
| major, minor = parsed | |
| supported_major, supported_minor = SUPPORTED_SCHEMA | |
| if major != supported_major or minor > supported_minor: | |
| raise IncompatibleArtifactError( | |
| f"artifact schema_version {declared} is not readable by this " | |
| f"extraction build (supports {supported_major}.{supported_minor}.x). " | |
| "Re-parse the document, or update the adapter to the new shape." | |
| ) | |
| if minor < supported_minor: | |
| logger.info( | |
| "artifact_schema_version_older", | |
| extra={"declared": str(declared), "supported": f"{supported_major}.{supported_minor}"}, | |
| ) | |
| def chunk_from_dict(raw: dict[str, Any], doc_id: str, ordinal: int = 0) -> Chunk: | |
| """Map one artifact item onto the internal chunk. | |
| `kind`/`is_tabular` are reconciled: the draft carries a `kind` discriminator | |
| while the prototype carried booleans. Either is accepted. | |
| """ | |
| kind = raw.get("kind") | |
| pages = _first(raw, _PAGES_KEYS) or [] | |
| page_start = _first(raw, _PAGE_KEYS) | |
| if page_start is None: | |
| page_start = min(pages) if pages else 0 | |
| page_end = _first(raw, _PAGE_END_KEYS) | |
| if page_end is None: | |
| page_end = max(pages) if pages else page_start | |
| return Chunk( | |
| chunk_id=raw.get("chunk_id") or f"{doc_id}::{ordinal:04d}", | |
| doc_id=raw.get("doc_id") or doc_id, | |
| text=raw["text"], # verbatim, never cleaned | |
| page_start=int(page_start), | |
| page_end=int(page_end), | |
| ordinal=int(raw.get("ordinal", ordinal)), | |
| section_no=raw.get("section_no"), | |
| heading=raw.get("heading"), | |
| # Same story as `latex` below: the parsing half has always emitted this | |
| # and extraction dropped it here, which left the document's own | |
| # structure unavailable to a branch that was inventing prose instead. | |
| heading_path=list(raw.get("heading_path") or []), | |
| has_formula=bool(raw.get("has_formula", kind == "equation")), | |
| is_tabular=bool(raw.get("is_tabular", kind == "table")), | |
| bold_spans=list(raw.get("bold_spans") or []), | |
| # The parsing half has always emitted this; extraction used to drop it on | |
| # the floor here, which left `formula_latex` unprovable — `text` carries | |
| # rendered prose, so a LaTeX claim had nothing to be checked against. | |
| latex=list(raw.get("latex") or []), | |
| # Same story a third time: the parsing half emits `table_html` and | |
| # extraction dropped it here, so a table reached the model ONLY as | |
| # pipe-linearised text with its row/column structure destroyed. It is the | |
| # haystack the formula guard needs for tables and the input a future | |
| # table branch needs; carrying it costs nothing. | |
| table_html=raw.get("table_html") or None, | |
| # New at 0.4.0 — see ChunkAsset. A pointer the adapter drops is a pointer | |
| # that does not exist, which is exactly what happened to `table_html`. | |
| assets=[_asset(a) for a in (raw.get("assets") or [])], | |
| referenced_by=list(raw.get("referenced_by") or []), | |
| ) | |
| def _asset(raw: dict[str, Any]) -> ChunkAsset: | |
| """Map one artifact asset onto the internal shape, tolerating absences.""" | |
| return ChunkAsset( | |
| asset_id=str(raw.get("asset_id") or ""), | |
| kind=str(raw.get("kind") or "figure"), | |
| caption=raw.get("caption") or None, | |
| description=raw.get("description") or None, | |
| page_no=raw.get("page_no"), | |
| storage_key=raw.get("storage_key") or None, | |
| ) | |
| def parsed_doc_from_artifact( | |
| artifact: Any, | |
| doc_id: str | None = None, | |
| source_ref: str = "", | |
| parser_name: str = "unknown", | |
| parser_version: str = "", | |
| ) -> ParsedDoc: | |
| """Build a `ParsedDoc` from either shape of the artifact. | |
| Accepts a bare `list[chunk]` (the draft's current shape) or a mapping with a | |
| `chunks` key (the shape proposed for the document-level envelope). When the | |
| envelope lands, its `content_hash`/`n_pages`/`version` are preferred over | |
| the values derived here. | |
| """ | |
| if hasattr(artifact, "model_dump"): # a ParsedDocument from the parsing half | |
| artifact = artifact.model_dump(mode="json") | |
| if isinstance(artifact, dict): | |
| items = artifact.get("chunks") or [] | |
| doc_id = doc_id or artifact.get("doc_id") | |
| source_ref = source_ref or artifact.get("source_path") or artifact.get("source_ref") or "" | |
| parser_name = artifact.get("parser_name") or parser_name | |
| parser_version = artifact.get("parser_version") or parser_version | |
| # The backend matters as much as the version: the same MinerU build can | |
| # emit different text from `pipeline` and `vlm`, so a shift in extraction | |
| # output has to be attributable to one or the other. | |
| backend = artifact.get("parser_backend") or "" | |
| if backend: | |
| parser_version = f"{parser_version}/{backend}" if parser_version else backend | |
| # Gate the shape BEFORE reading any of it (N6b). | |
| check_schema_version(artifact.get("schema_version")) | |
| declared_hash = artifact.get("content_hash") | |
| declared_pages = artifact.get("n_pages") | |
| else: | |
| # A bare chunk list carries no provenance at all. It stays loadable for | |
| # fixtures, but it arrives unlabelled and the CLI will refuse it. | |
| items = list(artifact) | |
| backend = "" | |
| declared_hash, declared_pages = None, None | |
| if not doc_id: | |
| doc_id = (items[0].get("doc_id") if items else None) or "unknown" | |
| chunks = [chunk_from_dict(raw, doc_id, i) for i, raw in enumerate(items)] | |
| pages = {p for c in chunks for p in (c.page_start, c.page_end)} | |
| return ParsedDoc( | |
| doc_id=doc_id, | |
| source_ref=source_ref, | |
| content_hash=declared_hash or content_hash(chunks), | |
| n_pages=int(declared_pages) if declared_pages else (max(pages) + 1 if pages else 0), | |
| chunks=chunks, | |
| parser_name=parser_name, | |
| parser_version=parser_version, | |
| parser_backend=backend, | |
| used_heading_split=any(c.section_no for c in chunks), | |
| ) | |
| def content_hash(chunks: list[Chunk]) -> str: | |
| """Stable hash of the chunk text, so a re-parse that changed nothing can be | |
| detected and the expensive stages skipped.""" | |
| blob = json.dumps([c.text for c in chunks], ensure_ascii=False).encode() | |
| return hashlib.sha256(blob).hexdigest()[:16] | |
| def _first(raw: dict[str, Any], keys: tuple[str, ...]) -> Any: | |
| for key in keys: | |
| if raw.get(key) is not None: | |
| return raw[key] | |
| return None | |