ishaq101's picture sofhiaazzhr's picture
/feat knowledge management (#20)
b68816f
Raw History Blame Contribute Delete
17 kB
"""Stage 3 — turn MinerU's flat item list into meaningful chunks.
Why this stage exists: MinerU emits a FLAT list of items, not meaningful units.
Those items have to be grouped back into per-section chunks.
MinerU marks headings with `text_level` (2 / 2.1 / 2.1.1 -> level 2 / 3 / 4) on
BOTH backends — checked against its source, `pipeline` and `vlm` run identical
logic. But that marker only appears when the document actually has detectable
headings:
- the BUMA standard (explicit numbered headings) -> a full hierarchy
- the McGraw-Hill handbook (a run of mid-chapter pages) -> almost none,
1 item out of 84, and that one a table caption
So heading availability is a property of the DOCUMENT, not of the backend. Hence
`text_level` is the primary signal, numbering patterns are the fallback, and the
result is allowed to come back empty.
What happens here:
- `page_number` is dropped (a page number is furniture, not content)
- `header` becomes chapter context rather than a section heading
(in the test documents every running header sits at bbox y~57-72, one per page)
- section headings come from `text_level`, falling back to a numbering pattern
("2.1.3 Title"), and are assembled into `heading_path` (the trail of parent
headings down to the chunk's own)
- `table`, `chart` and `image` each become their own chunk; `equation` attaches
to the running chunk and sets `has_formula`
⚠️ Text is NEVER tidied here — no rejoining lines, no whitespace normalisation.
See contracts.py for why.
"""
from __future__ import annotations
import hashlib
import json
import re
from functools import lru_cache
from pathlib import Path
from typing import Any
from .checks import trustworthy_vocabulary
from .contracts import Chunk
from .render import render_latex, render_table
# The modules whose code decides what a chunk CONTAINS. `parser_config` already
# fingerprints MinerU's settings, but nothing fingerprinted this half — so an
# artifact built before a normalisation fix and one built after were
# indistinguishable from their metadata. That is not hypothetical: on 2026-08-26
# a stale artifact was handed to the extraction half, term recall came back
# 0.7073 against a 0.8049 baseline, and it read as a regression in extraction's
# own work for a day. The only observable difference was the character count,
# and nobody had reason to check it.
#
# `contracts.py` is deliberately NOT here. The artifact's SHAPE is already
# versioned by `schema_version`, so hashing it too counts the same change twice —
# and it churns on edits that alter neither shape nor content. Moving a field's
# declaration order changed this hash once while producing byte-identical chunks,
# which is a false "your artifact is stale" and teaches people to ignore the
# signal. What this hash must cover is code that decides chunk CONTENT.
_CONTENT_MODULES = ("normalize.py", "render.py", "checks.py")
@lru_cache(maxsize=1)
def normaliser_version() -> str:
"""Fingerprint of the code that turns MinerU output into chunks.
A source hash, not a hand-maintained constant, and that is the whole point:
a version someone must remember to bump is a version that silently goes
stale, which is the failure being fixed here.
It therefore changes on edits that cannot alter output — a comment, a
docstring. That is deliberate. A false "this artifact is old" costs seconds:
rebuilding from the parse cache needs no GPU and produces identical bytes.
A missed "this artifact is old" costs a day. The noisy direction is the safe
one.
"""
here = Path(__file__).parent
h = hashlib.sha256()
for name in _CONTENT_MODULES:
h.update((here / name).read_bytes())
return h.hexdigest()[:12]
# "2.1.3 Title" / "2.1.3. Title" / "4 Title"
_NUMBER_PATTERN = re.compile(r"^(\d+(?:\.\d+)*)\.?\s+(\S.*)$")
# Headings are usually short. This threshold stops a paragraph that happens to
# start with a digit from being taken for one. The value is aligned with the
# calibrated constants in KNOWLEDGE_PIPELINE_CALIBRATION.md §4: a line longer
# than this is a sentence or a formula line, not a heading.
_MAX_HEADING_LEN = 90
# Page furniture, dropped on purpose. `footer` was previously dropped by
# ACCIDENT rather than decision: it has no `text` in the BUMA standard, so it
# fell out of the generic path unnoticed — but it does carry text in the
# McGraw-Hill handbook, where the same items leaked into chunks instead. Naming
# it here makes the two documents behave the same way, deliberately.
_DISCARDED = {"page_number", "footer"}
# Chunk size cap. Headings remain the primary boundary; this is only a guard for
# documents where no heading is detected at all — without it one chunk could
# swallow an entire document, which blunts evidence ranking and inflates the
# summary branch's token share. ~6000 characters is roughly 1,500 tokens.
MAX_CHUNK_CHARS = 6000
def _heading(item: dict[str, Any]) -> tuple[str | None, str | None, int | None]:
"""Return (section_no, heading, level) if this item is a section heading.
⭐ ONLY a NUMBERED heading opens a new section.
MinerU's `text_level` is not enough on its own: in the BUMA standard it also
marks "Keterangan:" and "Keterangan grafik:" as headings. Treating those as
section boundaries separates a legend from the figure it explains — and terms
like "Other Activity" and "Uncontrollable" disappear, despite sitting as
prose in a chunk that already passed the filter. That was the remaining
recall gap against the plain-text path.
So `text_level` decides the hierarchy LEVEL, while the numbering decides
whether a line is a section boundary at all. A `text_level` line without a
number stays part of the running chunk's content.
"""
if item.get("type") not in {"text", "title"}:
return None, None, None
text = (item.get("text") or "").strip()
if not text or len(text) > _MAX_HEADING_LEN:
return None, None, None
m = _NUMBER_PATTERN.match(text)
if not m or text.endswith((".", ":", ";")):
return None, None, None
level = item.get("text_level")
if level:
return m.group(1), m.group(2), int(level)
# With no text_level, the level is inferred from numbering depth:
# "2" -> 1, "2.1" -> 2, "2.1.1" -> 3
return m.group(1), m.group(2), m.group(1).count(".") + 1
def _table_text(item: dict[str, Any], vocabulary: frozenset[str] | None = None) -> str:
"""Caption + table body as readable text.
The raw HTML is NOT placed here — it is kept separately in
`Chunk.table_html`. See render.py for the measured reason.
"""
parts = list(item.get("table_caption") or [])
if item.get("table_body"):
parts.append(render_table(item["table_body"], vocabulary))
parts += list(item.get("table_footnote") or [])
return "\n\n".join(p for p in parts if p)
def _figure_text(item: dict[str, Any], kind: str) -> tuple[str, str | None]:
"""(verbatim text, model-written description) for one chart/image item.
`content` is DELIBERATELY separated from the text. On the `pipeline` backend
that field is always empty, which is why it went unnoticed — but `vlm` and
`hybrid --effort high` fill it with a description produced by image analysis.
Putting it in `text` would make model-written prose part of the haystack the
extraction span check searches, so a hallucination could pass the very check
built to catch it. A caption actually printed in the document is verbatim,
and stays in `text`.
"""
parts = list(item.get(f"{kind}_caption") or [])
parts += list(item.get(f"{kind}_footnote") or [])
description = (item.get("content") or "").strip() or None
return "\n\n".join(p for p in parts if p), description
def normalise(
items: list[dict[str, Any]],
doc_id: str,
page_vocabulary: dict[int, frozenset[str]] | None = None,
) -> list[Chunk]:
"""`page_vocabulary` is optional: the source PDF's words, per page.
Used only to restore the word boundaries lost when MinerU writes formulas one
character at a time (see `render.restore_word_boundaries`). Without it the
result is exactly as it was before.
"""
# The vocabulary is filtered first: a page whose text layer disagrees with
# MinerU's own reading is not fit to arbitrate, and is dropped here.
trusted = trustworthy_vocabulary(page_vocabulary, items) if page_vocabulary else {}
def vocabulary_for(page: int) -> frozenset[str] | None:
return trusted.get(page)
# Chapter context per page, taken from the running header
chapter_per_page: dict[int, str] = {}
for x in items:
if x.get("type") == "header" and (x.get("text") or "").strip():
chapter_per_page.setdefault(x.get("page_idx", 0), x["text"].strip())
chunks: list[Chunk] = []
running: Chunk | None = None
pieces: list[str] = []
# The stack of headings currently in force: [(level, heading_text), ...].
# A level-N heading closes every heading at level >= N before it.
stack: list[tuple[int, str]] = []
def push_heading(level: int, text: str) -> None:
while stack and stack[-1][0] >= level:
stack.pop()
stack.append((level, text))
def close() -> None:
nonlocal running, pieces
if running is not None:
running.text = "\n\n".join(pieces).strip()
if running.text or running.images:
running.page_idxs = sorted(set(running.page_idxs))
chunks.append(running)
running, pieces = None, []
def open_chunk(kind: str, page: int, section_no=None, heading=None) -> Chunk:
return Chunk(
chunk_id=f"{doc_id}::{len(chunks):04d}",
doc_id=doc_id, kind=kind, text="",
page_idx=page, page_idxs=[page],
section_no=section_no, heading=heading,
chapters=[chapter_per_page[page]] if page in chapter_per_page else [],
heading_path=[t for _, t in stack],
)
for i, item in enumerate(items):
kind = item.get("type")
if kind in _DISCARDED or kind == "header":
continue
page = item.get("page_idx", 0)
if kind == "table":
close()
c = open_chunk("table", page)
c.text = _table_text(item, vocabulary_for(page))
c.is_tabular = True
c.table_html = item.get("table_body") or None
c.source_items = [i]
c.bbox = item.get("bbox")
if item.get("img_path"):
c.images = [item["img_path"]]
if c.text or c.images:
chunks.append(c)
continue
# `image` sits alongside `chart`: different backends label the same
# picture differently — what `pipeline` calls a chart, `vlm` calls an
# image. Without this branch an `image` item falls through to the generic
# text path, which reads only `item["text"]`, and pictures have no such
# field — so BOTH its caption and its description vanish in silence.
if kind in {"chart", "image"}:
close()
c = open_chunk(kind, page)
c.text, c.generated_description = _figure_text(item, kind)
c.source_items = [i]
c.bbox = item.get("bbox")
if item.get("img_path"):
c.images = [item["img_path"]]
chunks.append(c)
continue
# `list` keeps its content in `list_items` and has NO `text` field — the
# same trap as `chart`/`image` above, and it cost more: the generic path
# reads `item["text"]`, finds nothing, and drops the entire block in
# silence. Measured on the BUMA standard, two dropped list blocks took
# seven gold terms with them (`Uncontrollable`, `Joint survey`,
# `Truck count`, `Mineplan`, `EWH`, `Fleet management`, `Controllable`)
# — 0.7073 recall against 0.8537 for the parser it was meant to beat.
#
# List content is prose inside its section, so it joins the running text
# chunk rather than opening its own. It deliberately skips heading
# detection: a short bullet can look like a heading, and letting one open
# a section would split a list away from the paragraph introducing it.
if kind == "list":
entries = [str(s).strip() for s in (item.get("list_items") or []) if str(s).strip()]
if not entries:
continue
if running is None:
running = open_chunk("text", page)
running.bbox = item.get("bbox")
running.source_items.append(i)
running.page_idxs.append(page)
pieces.append("\n".join(entries)) # verbatim, never tidied
continue
if kind == "equation":
latex = (item.get("text") or "").strip()
if running is None:
running = open_chunk("text", page)
running.has_formula = True
running.source_items.append(i)
running.page_idxs.append(page)
if item.get("img_path"):
running.images.append(item["img_path"])
if latex:
running.latex.append(latex) # raw, for the formula branch
rendered = render_latex(latex, vocabulary_for(page))
if rendered:
pieces.append(rendered) # prose, so NER can find it
continue
# everything else: text
text = (item.get("text") or "").strip()
if not text:
continue
number, heading, level = _heading(item)
if heading is not None:
# BREADCRUMB: many documents reprint their heading trail at the top
# of every page (BUMA repeats "2. PENJELASAN PARAMETER /
# 2.1. Production Parameter" on pp. 2-8). A heading ALREADY on the
# stack is the running section or one of its parents — not a new
# section. Without this rule the same section splits repeatedly and
# every downstream figure breaks with it.
if any(text == t for _, t in stack):
if running is not None:
running.page_idxs.append(page) # section continues, page range grows
continue
close()
# The heading is pushed BEFORE the chunk opens, so the chunk carries
# its own heading at the end of heading_path.
push_heading(level or 1, text)
running = open_chunk("text", page, section_no=number, heading=heading)
running.source_items = [i]
running.bbox = item.get("bbox")
# The heading line is NOT copied into `text` — it lives in the
# `heading` field, verbatim.
#
# The opposite was tried, because in the BUMA standard the heading
# names the term and the section body then opens "Adalah ..." without
# repeating it, so the chunk defining a term does not contain that
# term. But the extraction side already handles this: its span-check
# haystack is assembled as `heading + text`, and its ranker treats a
# heading as a mention at offset 0.
#
# Copying it into `text` actively hurts: the heading is counted twice
# in the haystack, and its occurrence becomes a real mention, so
# `mention_count` inflates — and that figure is what gets compared
# against the frozen baseline (169 mentions -> 66 clusters). The v2
# comparison would shift for a reason unrelated to quality.
continue
if running is None:
running = open_chunk("text", page)
running.bbox = item.get("bbox")
running.source_items.append(i)
running.page_idxs.append(page)
chapter = chapter_per_page.get(page)
if chapter and chapter not in running.chapters:
running.chapters.append(chapter)
pieces.append(text) # verbatim, never tidied
# Size guard: only ever relevant for documents with no detected heading.
# The chunk is cut at an item boundary, so the text stays verbatim.
if sum(len(x) for x in pieces) >= MAX_CHUNK_CHARS:
continued_from = running
close()
running = open_chunk("text", page,
section_no=continued_from.section_no,
heading=continued_from.heading)
close()
return chunks
def normalise_from_file(
content_list: Path,
doc_id: str,
page_vocabulary: dict[int, frozenset[str]] | None = None,
) -> list[Chunk]:
items = json.loads(content_list.read_text(encoding="utf-8"))
return normalise(items, doc_id, page_vocabulary)