"""Stage 3 — turn MinerU's flat item list into meaningful chunks. Why this stage exists: MinerU emits a FLAT list of items, not meaningful units. Those items have to be grouped back into per-section chunks. MinerU marks headings with `text_level` (2 / 2.1 / 2.1.1 -> level 2 / 3 / 4) on BOTH backends — checked against its source, `pipeline` and `vlm` run identical logic. But that marker only appears when the document actually has detectable headings: - the BUMA standard (explicit numbered headings) -> a full hierarchy - the McGraw-Hill handbook (a run of mid-chapter pages) -> almost none, 1 item out of 84, and that one a table caption So heading availability is a property of the DOCUMENT, not of the backend. Hence `text_level` is the primary signal, numbering patterns are the fallback, and the result is allowed to come back empty. What happens here: - `page_number` is dropped (a page number is furniture, not content) - `header` becomes chapter context rather than a section heading (in the test documents every running header sits at bbox y~57-72, one per page) - section headings come from `text_level`, falling back to a numbering pattern ("2.1.3 Title"), and are assembled into `heading_path` (the trail of parent headings down to the chunk's own) - `table`, `chart` and `image` each become their own chunk; `equation` attaches to the running chunk and sets `has_formula` ⚠️ Text is NEVER tidied here — no rejoining lines, no whitespace normalisation. See contracts.py for why. """ from __future__ import annotations import hashlib import json import re from functools import lru_cache from pathlib import Path from typing import Any from .checks import trustworthy_vocabulary from .contracts import Chunk from .render import render_latex, render_table # The modules whose code decides what a chunk CONTAINS. `parser_config` already # fingerprints MinerU's settings, but nothing fingerprinted this half — so an # artifact built before a normalisation fix and one built after were # indistinguishable from their metadata. That is not hypothetical: on 2026-08-26 # a stale artifact was handed to the extraction half, term recall came back # 0.7073 against a 0.8049 baseline, and it read as a regression in extraction's # own work for a day. The only observable difference was the character count, # and nobody had reason to check it. # # `contracts.py` is deliberately NOT here. The artifact's SHAPE is already # versioned by `schema_version`, so hashing it too counts the same change twice — # and it churns on edits that alter neither shape nor content. Moving a field's # declaration order changed this hash once while producing byte-identical chunks, # which is a false "your artifact is stale" and teaches people to ignore the # signal. What this hash must cover is code that decides chunk CONTENT. _CONTENT_MODULES = ("normalize.py", "render.py", "checks.py") @lru_cache(maxsize=1) def normaliser_version() -> str: """Fingerprint of the code that turns MinerU output into chunks. A source hash, not a hand-maintained constant, and that is the whole point: a version someone must remember to bump is a version that silently goes stale, which is the failure being fixed here. It therefore changes on edits that cannot alter output — a comment, a docstring. That is deliberate. A false "this artifact is old" costs seconds: rebuilding from the parse cache needs no GPU and produces identical bytes. A missed "this artifact is old" costs a day. The noisy direction is the safe one. """ here = Path(__file__).parent h = hashlib.sha256() for name in _CONTENT_MODULES: h.update((here / name).read_bytes()) return h.hexdigest()[:12] # "2.1.3 Title" / "2.1.3. Title" / "4 Title" _NUMBER_PATTERN = re.compile(r"^(\d+(?:\.\d+)*)\.?\s+(\S.*)$") # Headings are usually short. This threshold stops a paragraph that happens to # start with a digit from being taken for one. The value is aligned with the # calibrated constants in KNOWLEDGE_PIPELINE_CALIBRATION.md §4: a line longer # than this is a sentence or a formula line, not a heading. _MAX_HEADING_LEN = 90 # Page furniture, dropped on purpose. `footer` was previously dropped by # ACCIDENT rather than decision: it has no `text` in the BUMA standard, so it # fell out of the generic path unnoticed — but it does carry text in the # McGraw-Hill handbook, where the same items leaked into chunks instead. Naming # it here makes the two documents behave the same way, deliberately. _DISCARDED = {"page_number", "footer"} # Chunk size cap. Headings remain the primary boundary; this is only a guard for # documents where no heading is detected at all — without it one chunk could # swallow an entire document, which blunts evidence ranking and inflates the # summary branch's token share. ~6000 characters is roughly 1,500 tokens. MAX_CHUNK_CHARS = 6000 def _heading(item: dict[str, Any]) -> tuple[str | None, str | None, int | None]: """Return (section_no, heading, level) if this item is a section heading. ⭐ ONLY a NUMBERED heading opens a new section. MinerU's `text_level` is not enough on its own: in the BUMA standard it also marks "Keterangan:" and "Keterangan grafik:" as headings. Treating those as section boundaries separates a legend from the figure it explains — and terms like "Other Activity" and "Uncontrollable" disappear, despite sitting as prose in a chunk that already passed the filter. That was the remaining recall gap against the plain-text path. So `text_level` decides the hierarchy LEVEL, while the numbering decides whether a line is a section boundary at all. A `text_level` line without a number stays part of the running chunk's content. """ if item.get("type") not in {"text", "title"}: return None, None, None text = (item.get("text") or "").strip() if not text or len(text) > _MAX_HEADING_LEN: return None, None, None m = _NUMBER_PATTERN.match(text) if not m or text.endswith((".", ":", ";")): return None, None, None level = item.get("text_level") if level: return m.group(1), m.group(2), int(level) # With no text_level, the level is inferred from numbering depth: # "2" -> 1, "2.1" -> 2, "2.1.1" -> 3 return m.group(1), m.group(2), m.group(1).count(".") + 1 def _table_text(item: dict[str, Any], vocabulary: frozenset[str] | None = None) -> str: """Caption + table body as readable text. The raw HTML is NOT placed here — it is kept separately in `Chunk.table_html`. See render.py for the measured reason. """ parts = list(item.get("table_caption") or []) if item.get("table_body"): parts.append(render_table(item["table_body"], vocabulary)) parts += list(item.get("table_footnote") or []) return "\n\n".join(p for p in parts if p) def _figure_text(item: dict[str, Any], kind: str) -> tuple[str, str | None]: """(verbatim text, model-written description) for one chart/image item. `content` is DELIBERATELY separated from the text. On the `pipeline` backend that field is always empty, which is why it went unnoticed — but `vlm` and `hybrid --effort high` fill it with a description produced by image analysis. Putting it in `text` would make model-written prose part of the haystack the extraction span check searches, so a hallucination could pass the very check built to catch it. A caption actually printed in the document is verbatim, and stays in `text`. """ parts = list(item.get(f"{kind}_caption") or []) parts += list(item.get(f"{kind}_footnote") or []) description = (item.get("content") or "").strip() or None return "\n\n".join(p for p in parts if p), description def normalise( items: list[dict[str, Any]], doc_id: str, page_vocabulary: dict[int, frozenset[str]] | None = None, ) -> list[Chunk]: """`page_vocabulary` is optional: the source PDF's words, per page. Used only to restore the word boundaries lost when MinerU writes formulas one character at a time (see `render.restore_word_boundaries`). Without it the result is exactly as it was before. """ # The vocabulary is filtered first: a page whose text layer disagrees with # MinerU's own reading is not fit to arbitrate, and is dropped here. trusted = trustworthy_vocabulary(page_vocabulary, items) if page_vocabulary else {} def vocabulary_for(page: int) -> frozenset[str] | None: return trusted.get(page) # Chapter context per page, taken from the running header chapter_per_page: dict[int, str] = {} for x in items: if x.get("type") == "header" and (x.get("text") or "").strip(): chapter_per_page.setdefault(x.get("page_idx", 0), x["text"].strip()) chunks: list[Chunk] = [] running: Chunk | None = None pieces: list[str] = [] # The stack of headings currently in force: [(level, heading_text), ...]. # A level-N heading closes every heading at level >= N before it. stack: list[tuple[int, str]] = [] def push_heading(level: int, text: str) -> None: while stack and stack[-1][0] >= level: stack.pop() stack.append((level, text)) def close() -> None: nonlocal running, pieces if running is not None: running.text = "\n\n".join(pieces).strip() if running.text or running.images: running.page_idxs = sorted(set(running.page_idxs)) chunks.append(running) running, pieces = None, [] def open_chunk(kind: str, page: int, section_no=None, heading=None) -> Chunk: return Chunk( chunk_id=f"{doc_id}::{len(chunks):04d}", doc_id=doc_id, kind=kind, text="", page_idx=page, page_idxs=[page], section_no=section_no, heading=heading, chapters=[chapter_per_page[page]] if page in chapter_per_page else [], heading_path=[t for _, t in stack], ) for i, item in enumerate(items): kind = item.get("type") if kind in _DISCARDED or kind == "header": continue page = item.get("page_idx", 0) if kind == "table": close() c = open_chunk("table", page) c.text = _table_text(item, vocabulary_for(page)) c.is_tabular = True c.table_html = item.get("table_body") or None c.source_items = [i] c.bbox = item.get("bbox") if item.get("img_path"): c.images = [item["img_path"]] if c.text or c.images: chunks.append(c) continue # `image` sits alongside `chart`: different backends label the same # picture differently — what `pipeline` calls a chart, `vlm` calls an # image. Without this branch an `image` item falls through to the generic # text path, which reads only `item["text"]`, and pictures have no such # field — so BOTH its caption and its description vanish in silence. if kind in {"chart", "image"}: close() c = open_chunk(kind, page) c.text, c.generated_description = _figure_text(item, kind) c.source_items = [i] c.bbox = item.get("bbox") if item.get("img_path"): c.images = [item["img_path"]] chunks.append(c) continue # `list` keeps its content in `list_items` and has NO `text` field — the # same trap as `chart`/`image` above, and it cost more: the generic path # reads `item["text"]`, finds nothing, and drops the entire block in # silence. Measured on the BUMA standard, two dropped list blocks took # seven gold terms with them (`Uncontrollable`, `Joint survey`, # `Truck count`, `Mineplan`, `EWH`, `Fleet management`, `Controllable`) # — 0.7073 recall against 0.8537 for the parser it was meant to beat. # # List content is prose inside its section, so it joins the running text # chunk rather than opening its own. It deliberately skips heading # detection: a short bullet can look like a heading, and letting one open # a section would split a list away from the paragraph introducing it. if kind == "list": entries = [str(s).strip() for s in (item.get("list_items") or []) if str(s).strip()] if not entries: continue if running is None: running = open_chunk("text", page) running.bbox = item.get("bbox") running.source_items.append(i) running.page_idxs.append(page) pieces.append("\n".join(entries)) # verbatim, never tidied continue if kind == "equation": latex = (item.get("text") or "").strip() if running is None: running = open_chunk("text", page) running.has_formula = True running.source_items.append(i) running.page_idxs.append(page) if item.get("img_path"): running.images.append(item["img_path"]) if latex: running.latex.append(latex) # raw, for the formula branch rendered = render_latex(latex, vocabulary_for(page)) if rendered: pieces.append(rendered) # prose, so NER can find it continue # everything else: text text = (item.get("text") or "").strip() if not text: continue number, heading, level = _heading(item) if heading is not None: # BREADCRUMB: many documents reprint their heading trail at the top # of every page (BUMA repeats "2. PENJELASAN PARAMETER / # 2.1. Production Parameter" on pp. 2-8). A heading ALREADY on the # stack is the running section or one of its parents — not a new # section. Without this rule the same section splits repeatedly and # every downstream figure breaks with it. if any(text == t for _, t in stack): if running is not None: running.page_idxs.append(page) # section continues, page range grows continue close() # The heading is pushed BEFORE the chunk opens, so the chunk carries # its own heading at the end of heading_path. push_heading(level or 1, text) running = open_chunk("text", page, section_no=number, heading=heading) running.source_items = [i] running.bbox = item.get("bbox") # The heading line is NOT copied into `text` — it lives in the # `heading` field, verbatim. # # The opposite was tried, because in the BUMA standard the heading # names the term and the section body then opens "Adalah ..." without # repeating it, so the chunk defining a term does not contain that # term. But the extraction side already handles this: its span-check # haystack is assembled as `heading + text`, and its ranker treats a # heading as a mention at offset 0. # # Copying it into `text` actively hurts: the heading is counted twice # in the haystack, and its occurrence becomes a real mention, so # `mention_count` inflates — and that figure is what gets compared # against the frozen baseline (169 mentions -> 66 clusters). The v2 # comparison would shift for a reason unrelated to quality. continue if running is None: running = open_chunk("text", page) running.bbox = item.get("bbox") running.source_items.append(i) running.page_idxs.append(page) chapter = chapter_per_page.get(page) if chapter and chapter not in running.chapters: running.chapters.append(chapter) pieces.append(text) # verbatim, never tidied # Size guard: only ever relevant for documents with no detected heading. # The chunk is cut at an item boundary, so the text stays verbatim. if sum(len(x) for x in pieces) >= MAX_CHUNK_CHARS: continued_from = running close() running = open_chunk("text", page, section_no=continued_from.section_no, heading=continued_from.heading) close() return chunks def normalise_from_file( content_list: Path, doc_id: str, page_vocabulary: dict[int, frozenset[str]] | None = None, ) -> list[Chunk]: items = json.loads(content_list.read_text(encoding="utf-8")) return normalise(items, doc_id, page_vocabulary)