ishaq101's picture
/fix parsing and term extract (#21)
f07443e
Raw History Blame Contribute Delete
11.3 kB
"""Seam adapter: parsed-document artifact → the pipeline's internal `Chunk`.
**This is the only file that knows the seam's shape.** Every stage downstream
depends on `models.Chunk` alone, so when the artifact contract settles with
Sofhia the change lands here and nowhere else.
The seam is still under discussion (KNOWLEDGE_PIPELINE_TODO.md §3), so this
reads defensively: it accepts either the draft's field names or the prototype's,
takes plain dicts, and tolerates missing optional structure. It deliberately
does **not** accept a file path — extraction never opens a document. That
constraint is the point of the split, not an implementation detail.
Two things it must never do:
- reflow, strip or whitespace-normalise `text`. Span validation locates quoted
spans literally inside it; cleaning the text makes the lookup fail and the
field go silently null, which presents as a bad model.
- infer a page number it was not given. A wrong page sends the reviewer to the
wrong part of the document, which is worse than no page at all.
"""
from __future__ import annotations
import hashlib
import json
from typing import Any
from src.middlewares.logging import get_logger
from .models import Chunk, ChunkAsset, ParsedDoc
logger = get_logger("knowledge_adapter")
# Field names accepted for the same concept. `page_idx`/`page_idxs` are the
# parsing half's contract; the rest are earlier drafts and the prototype's shape,
# kept so old fixtures still load.
#
# Page numbers are 0-BASED throughout, exactly as the parser reports them. No
# conversion happens anywhere in this pipeline: converting to 1-based is the
# review UI's job, done once at display time. An off-by-one here would be
# invisible until an expert opened the wrong page.
_PAGE_KEYS = ("page_idx", "page_start", "page")
_PAGES_KEYS = ("page_idxs", "pages", "page_list")
_PAGE_END_KEYS = ("page_idx_end", "page_end")
# ── Seam version gate (N6b) ──────────────────────────────────────────────
#
# The artifact's SHAPE version, as `major.minor`. Parsing owns the contract and
# bumps `SCHEMA_VERSION` in `knowledge_parsing/contracts.py`; this is the range
# extraction knows how to read.
#
# Deliberately a literal rather than an import: extraction must never depend on
# the parsing package at runtime — that independence is the whole point of the
# seam. Drift is caught instead by a local seam test that imports
# `knowledge_parsing.contracts` directly and asserts the two agree, so a bump on
# either side fails loudly in tests rather than silently at read time.
SUPPORTED_SCHEMA = (0, 4)
class IncompatibleArtifactError(ValueError):
"""The artifact declares a shape this code cannot read.
Raised rather than degraded on purpose. Every other seam in this repo
tolerates missing structure, because a chunk without a heading is still
usable — but a shape we do not understand is not a smaller version of the
right input, it is a different input. Reading it anyway produces entries
that look fine and are wrong, which is the single most expensive failure
this pipeline can have.
"""
def _parse_version(declared: str) -> tuple[int, int] | None:
parts = str(declared).split(".")
try:
return int(parts[0]), int(parts[1])
except (IndexError, ValueError):
return None
def check_schema_version(declared: Any) -> None:
"""Gate one artifact's declared `schema_version`. Raises or returns None.
- **Major mismatch → reject.** A major bump is by definition breaking.
- **Newer minor → reject.** The artifact carries structure this code has
never seen; guessing at it is how a silent misread starts.
- **Older minor, or any patch → accept.** Minor bumps have been additive by
convention (0.3.0 → 0.3.3 added `source_title`, `normaliser_version` and
`page_no`), and the committed eval fixture is deliberately an older minor
than the current parser, so rejecting those would break the baseline
measurement for no safety gain.
- **Absent → accept with a warning.** Bare chunk lists and pre-envelope
fixtures carry no version; the CLI already refuses unlabelled artifacts
where labelling matters.
"""
if declared in (None, ""):
logger.warning(
"artifact_schema_version_missing",
extra={"supported": ".".join(map(str, SUPPORTED_SCHEMA))},
)
return
parsed = _parse_version(declared)
if parsed is None:
raise IncompatibleArtifactError(
f"artifact schema_version {declared!r} is not a major.minor version"
)
major, minor = parsed
supported_major, supported_minor = SUPPORTED_SCHEMA
if major != supported_major or minor > supported_minor:
raise IncompatibleArtifactError(
f"artifact schema_version {declared} is not readable by this "
f"extraction build (supports {supported_major}.{supported_minor}.x). "
"Re-parse the document, or update the adapter to the new shape."
)
if minor < supported_minor:
logger.info(
"artifact_schema_version_older",
extra={"declared": str(declared), "supported": f"{supported_major}.{supported_minor}"},
)
def chunk_from_dict(raw: dict[str, Any], doc_id: str, ordinal: int = 0) -> Chunk:
"""Map one artifact item onto the internal chunk.
`kind`/`is_tabular` are reconciled: the draft carries a `kind` discriminator
while the prototype carried booleans. Either is accepted.
"""
kind = raw.get("kind")
pages = _first(raw, _PAGES_KEYS) or []
page_start = _first(raw, _PAGE_KEYS)
if page_start is None:
page_start = min(pages) if pages else 0
page_end = _first(raw, _PAGE_END_KEYS)
if page_end is None:
page_end = max(pages) if pages else page_start
return Chunk(
chunk_id=raw.get("chunk_id") or f"{doc_id}::{ordinal:04d}",
doc_id=raw.get("doc_id") or doc_id,
text=raw["text"], # verbatim, never cleaned
page_start=int(page_start),
page_end=int(page_end),
ordinal=int(raw.get("ordinal", ordinal)),
section_no=raw.get("section_no"),
heading=raw.get("heading"),
# Same story as `latex` below: the parsing half has always emitted this
# and extraction dropped it here, which left the document's own
# structure unavailable to a branch that was inventing prose instead.
heading_path=list(raw.get("heading_path") or []),
has_formula=bool(raw.get("has_formula", kind == "equation")),
is_tabular=bool(raw.get("is_tabular", kind == "table")),
bold_spans=list(raw.get("bold_spans") or []),
# The parsing half has always emitted this; extraction used to drop it on
# the floor here, which left `formula_latex` unprovable — `text` carries
# rendered prose, so a LaTeX claim had nothing to be checked against.
latex=list(raw.get("latex") or []),
# Same story a third time: the parsing half emits `table_html` and
# extraction dropped it here, so a table reached the model ONLY as
# pipe-linearised text with its row/column structure destroyed. It is the
# haystack the formula guard needs for tables and the input a future
# table branch needs; carrying it costs nothing.
table_html=raw.get("table_html") or None,
# New at 0.4.0 — see ChunkAsset. A pointer the adapter drops is a pointer
# that does not exist, which is exactly what happened to `table_html`.
assets=[_asset(a) for a in (raw.get("assets") or [])],
referenced_by=list(raw.get("referenced_by") or []),
)
def _asset(raw: dict[str, Any]) -> ChunkAsset:
"""Map one artifact asset onto the internal shape, tolerating absences."""
return ChunkAsset(
asset_id=str(raw.get("asset_id") or ""),
kind=str(raw.get("kind") or "figure"),
caption=raw.get("caption") or None,
description=raw.get("description") or None,
page_no=raw.get("page_no"),
storage_key=raw.get("storage_key") or None,
)
def parsed_doc_from_artifact(
artifact: Any,
doc_id: str | None = None,
source_ref: str = "",
parser_name: str = "unknown",
parser_version: str = "",
) -> ParsedDoc:
"""Build a `ParsedDoc` from either shape of the artifact.
Accepts a bare `list[chunk]` (the draft's current shape) or a mapping with a
`chunks` key (the shape proposed for the document-level envelope). When the
envelope lands, its `content_hash`/`n_pages`/`version` are preferred over
the values derived here.
"""
if hasattr(artifact, "model_dump"): # a ParsedDocument from the parsing half
artifact = artifact.model_dump(mode="json")
if isinstance(artifact, dict):
items = artifact.get("chunks") or []
doc_id = doc_id or artifact.get("doc_id")
source_ref = source_ref or artifact.get("source_path") or artifact.get("source_ref") or ""
parser_name = artifact.get("parser_name") or parser_name
parser_version = artifact.get("parser_version") or parser_version
# The backend matters as much as the version: the same MinerU build can
# emit different text from `pipeline` and `vlm`, so a shift in extraction
# output has to be attributable to one or the other.
backend = artifact.get("parser_backend") or ""
if backend:
parser_version = f"{parser_version}/{backend}" if parser_version else backend
# Gate the shape BEFORE reading any of it (N6b).
check_schema_version(artifact.get("schema_version"))
declared_hash = artifact.get("content_hash")
declared_pages = artifact.get("n_pages")
else:
# A bare chunk list carries no provenance at all. It stays loadable for
# fixtures, but it arrives unlabelled and the CLI will refuse it.
items = list(artifact)
backend = ""
declared_hash, declared_pages = None, None
if not doc_id:
doc_id = (items[0].get("doc_id") if items else None) or "unknown"
chunks = [chunk_from_dict(raw, doc_id, i) for i, raw in enumerate(items)]
pages = {p for c in chunks for p in (c.page_start, c.page_end)}
return ParsedDoc(
doc_id=doc_id,
source_ref=source_ref,
content_hash=declared_hash or content_hash(chunks),
n_pages=int(declared_pages) if declared_pages else (max(pages) + 1 if pages else 0),
chunks=chunks,
parser_name=parser_name,
parser_version=parser_version,
parser_backend=backend,
used_heading_split=any(c.section_no for c in chunks),
)
def content_hash(chunks: list[Chunk]) -> str:
"""Stable hash of the chunk text, so a re-parse that changed nothing can be
detected and the expensive stages skipped."""
blob = json.dumps([c.text for c in chunks], ensure_ascii=False).encode()
return hashlib.sha256(blob).hexdigest()[:16]
def _first(raw: dict[str, Any], keys: tuple[str, ...]) -> Any:
for key in keys:
if raw.get(key) is not None:
return raw[key]
return None