Download src/knowledge_parsing/normalize.py from DataEyond/Agentic-Service-Data-Eyond-Catalog: direct link, hf CLI and curl.
- Browser
- Download file 17 kB
-
https://huggingface.co/spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/resolve/main/src/knowledge_parsing/normalize.py
- Command line
-
hf download hf://spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/src/knowledge_parsing/normalize.py
-
curl -L -o normalize.py https://huggingface.co/spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/resolve/main/src/knowledge_parsing/normalize.py
17 kB
| """Stage 3 — turn MinerU's flat item list into meaningful chunks. | |
| Why this stage exists: MinerU emits a FLAT list of items, not meaningful units. | |
| Those items have to be grouped back into per-section chunks. | |
| MinerU marks headings with `text_level` (2 / 2.1 / 2.1.1 -> level 2 / 3 / 4) on | |
| BOTH backends — checked against its source, `pipeline` and `vlm` run identical | |
| logic. But that marker only appears when the document actually has detectable | |
| headings: | |
| - the BUMA standard (explicit numbered headings) -> a full hierarchy | |
| - the McGraw-Hill handbook (a run of mid-chapter pages) -> almost none, | |
| 1 item out of 84, and that one a table caption | |
| So heading availability is a property of the DOCUMENT, not of the backend. Hence | |
| `text_level` is the primary signal, numbering patterns are the fallback, and the | |
| result is allowed to come back empty. | |
| What happens here: | |
| - `page_number` is dropped (a page number is furniture, not content) | |
| - `header` becomes chapter context rather than a section heading | |
| (in the test documents every running header sits at bbox y~57-72, one per page) | |
| - section headings come from `text_level`, falling back to a numbering pattern | |
| ("2.1.3 Title"), and are assembled into `heading_path` (the trail of parent | |
| headings down to the chunk's own) | |
| - `table`, `chart` and `image` each become their own chunk; `equation` attaches | |
| to the running chunk and sets `has_formula` | |
| ⚠️ Text is NEVER tidied here — no rejoining lines, no whitespace normalisation. | |
| See contracts.py for why. | |
| """ | |
| from __future__ import annotations | |
| import hashlib | |
| import json | |
| import re | |
| from functools import lru_cache | |
| from pathlib import Path | |
| from typing import Any | |
| from .checks import trustworthy_vocabulary | |
| from .contracts import Chunk | |
| from .render import render_latex, render_table | |
| # The modules whose code decides what a chunk CONTAINS. `parser_config` already | |
| # fingerprints MinerU's settings, but nothing fingerprinted this half — so an | |
| # artifact built before a normalisation fix and one built after were | |
| # indistinguishable from their metadata. That is not hypothetical: on 2026-08-26 | |
| # a stale artifact was handed to the extraction half, term recall came back | |
| # 0.7073 against a 0.8049 baseline, and it read as a regression in extraction's | |
| # own work for a day. The only observable difference was the character count, | |
| # and nobody had reason to check it. | |
| # | |
| # `contracts.py` is deliberately NOT here. The artifact's SHAPE is already | |
| # versioned by `schema_version`, so hashing it too counts the same change twice — | |
| # and it churns on edits that alter neither shape nor content. Moving a field's | |
| # declaration order changed this hash once while producing byte-identical chunks, | |
| # which is a false "your artifact is stale" and teaches people to ignore the | |
| # signal. What this hash must cover is code that decides chunk CONTENT. | |
| _CONTENT_MODULES = ("normalize.py", "render.py", "checks.py") | |
| def normaliser_version() -> str: | |
| """Fingerprint of the code that turns MinerU output into chunks. | |
| A source hash, not a hand-maintained constant, and that is the whole point: | |
| a version someone must remember to bump is a version that silently goes | |
| stale, which is the failure being fixed here. | |
| It therefore changes on edits that cannot alter output — a comment, a | |
| docstring. That is deliberate. A false "this artifact is old" costs seconds: | |
| rebuilding from the parse cache needs no GPU and produces identical bytes. | |
| A missed "this artifact is old" costs a day. The noisy direction is the safe | |
| one. | |
| """ | |
| here = Path(__file__).parent | |
| h = hashlib.sha256() | |
| for name in _CONTENT_MODULES: | |
| h.update((here / name).read_bytes()) | |
| return h.hexdigest()[:12] | |
| # "2.1.3 Title" / "2.1.3. Title" / "4 Title" | |
| _NUMBER_PATTERN = re.compile(r"^(\d+(?:\.\d+)*)\.?\s+(\S.*)$") | |
| # Headings are usually short. This threshold stops a paragraph that happens to | |
| # start with a digit from being taken for one. The value is aligned with the | |
| # calibrated constants in KNOWLEDGE_PIPELINE_CALIBRATION.md §4: a line longer | |
| # than this is a sentence or a formula line, not a heading. | |
| _MAX_HEADING_LEN = 90 | |
| # Page furniture, dropped on purpose. `footer` was previously dropped by | |
| # ACCIDENT rather than decision: it has no `text` in the BUMA standard, so it | |
| # fell out of the generic path unnoticed — but it does carry text in the | |
| # McGraw-Hill handbook, where the same items leaked into chunks instead. Naming | |
| # it here makes the two documents behave the same way, deliberately. | |
| _DISCARDED = {"page_number", "footer"} | |
| # Chunk size cap. Headings remain the primary boundary; this is only a guard for | |
| # documents where no heading is detected at all — without it one chunk could | |
| # swallow an entire document, which blunts evidence ranking and inflates the | |
| # summary branch's token share. ~6000 characters is roughly 1,500 tokens. | |
| MAX_CHUNK_CHARS = 6000 | |
| def _heading(item: dict[str, Any]) -> tuple[str | None, str | None, int | None]: | |
| """Return (section_no, heading, level) if this item is a section heading. | |
| ⭐ ONLY a NUMBERED heading opens a new section. | |
| MinerU's `text_level` is not enough on its own: in the BUMA standard it also | |
| marks "Keterangan:" and "Keterangan grafik:" as headings. Treating those as | |
| section boundaries separates a legend from the figure it explains — and terms | |
| like "Other Activity" and "Uncontrollable" disappear, despite sitting as | |
| prose in a chunk that already passed the filter. That was the remaining | |
| recall gap against the plain-text path. | |
| So `text_level` decides the hierarchy LEVEL, while the numbering decides | |
| whether a line is a section boundary at all. A `text_level` line without a | |
| number stays part of the running chunk's content. | |
| """ | |
| if item.get("type") not in {"text", "title"}: | |
| return None, None, None | |
| text = (item.get("text") or "").strip() | |
| if not text or len(text) > _MAX_HEADING_LEN: | |
| return None, None, None | |
| m = _NUMBER_PATTERN.match(text) | |
| if not m or text.endswith((".", ":", ";")): | |
| return None, None, None | |
| level = item.get("text_level") | |
| if level: | |
| return m.group(1), m.group(2), int(level) | |
| # With no text_level, the level is inferred from numbering depth: | |
| # "2" -> 1, "2.1" -> 2, "2.1.1" -> 3 | |
| return m.group(1), m.group(2), m.group(1).count(".") + 1 | |
| def _table_text(item: dict[str, Any], vocabulary: frozenset[str] | None = None) -> str: | |
| """Caption + table body as readable text. | |
| The raw HTML is NOT placed here — it is kept separately in | |
| `Chunk.table_html`. See render.py for the measured reason. | |
| """ | |
| parts = list(item.get("table_caption") or []) | |
| if item.get("table_body"): | |
| parts.append(render_table(item["table_body"], vocabulary)) | |
| parts += list(item.get("table_footnote") or []) | |
| return "\n\n".join(p for p in parts if p) | |
| def _figure_text(item: dict[str, Any], kind: str) -> tuple[str, str | None]: | |
| """(verbatim text, model-written description) for one chart/image item. | |
| `content` is DELIBERATELY separated from the text. On the `pipeline` backend | |
| that field is always empty, which is why it went unnoticed — but `vlm` and | |
| `hybrid --effort high` fill it with a description produced by image analysis. | |
| Putting it in `text` would make model-written prose part of the haystack the | |
| extraction span check searches, so a hallucination could pass the very check | |
| built to catch it. A caption actually printed in the document is verbatim, | |
| and stays in `text`. | |
| """ | |
| parts = list(item.get(f"{kind}_caption") or []) | |
| parts += list(item.get(f"{kind}_footnote") or []) | |
| description = (item.get("content") or "").strip() or None | |
| return "\n\n".join(p for p in parts if p), description | |
| def normalise( | |
| items: list[dict[str, Any]], | |
| doc_id: str, | |
| page_vocabulary: dict[int, frozenset[str]] | None = None, | |
| ) -> list[Chunk]: | |
| """`page_vocabulary` is optional: the source PDF's words, per page. | |
| Used only to restore the word boundaries lost when MinerU writes formulas one | |
| character at a time (see `render.restore_word_boundaries`). Without it the | |
| result is exactly as it was before. | |
| """ | |
| # The vocabulary is filtered first: a page whose text layer disagrees with | |
| # MinerU's own reading is not fit to arbitrate, and is dropped here. | |
| trusted = trustworthy_vocabulary(page_vocabulary, items) if page_vocabulary else {} | |
| def vocabulary_for(page: int) -> frozenset[str] | None: | |
| return trusted.get(page) | |
| # Chapter context per page, taken from the running header | |
| chapter_per_page: dict[int, str] = {} | |
| for x in items: | |
| if x.get("type") == "header" and (x.get("text") or "").strip(): | |
| chapter_per_page.setdefault(x.get("page_idx", 0), x["text"].strip()) | |
| chunks: list[Chunk] = [] | |
| running: Chunk | None = None | |
| pieces: list[str] = [] | |
| # The stack of headings currently in force: [(level, heading_text), ...]. | |
| # A level-N heading closes every heading at level >= N before it. | |
| stack: list[tuple[int, str]] = [] | |
| def push_heading(level: int, text: str) -> None: | |
| while stack and stack[-1][0] >= level: | |
| stack.pop() | |
| stack.append((level, text)) | |
| def close() -> None: | |
| nonlocal running, pieces | |
| if running is not None: | |
| running.text = "\n\n".join(pieces).strip() | |
| if running.text or running.images: | |
| running.page_idxs = sorted(set(running.page_idxs)) | |
| chunks.append(running) | |
| running, pieces = None, [] | |
| def open_chunk(kind: str, page: int, section_no=None, heading=None) -> Chunk: | |
| return Chunk( | |
| chunk_id=f"{doc_id}::{len(chunks):04d}", | |
| doc_id=doc_id, kind=kind, text="", | |
| page_idx=page, page_idxs=[page], | |
| section_no=section_no, heading=heading, | |
| chapters=[chapter_per_page[page]] if page in chapter_per_page else [], | |
| heading_path=[t for _, t in stack], | |
| ) | |
| for i, item in enumerate(items): | |
| kind = item.get("type") | |
| if kind in _DISCARDED or kind == "header": | |
| continue | |
| page = item.get("page_idx", 0) | |
| if kind == "table": | |
| close() | |
| c = open_chunk("table", page) | |
| c.text = _table_text(item, vocabulary_for(page)) | |
| c.is_tabular = True | |
| c.table_html = item.get("table_body") or None | |
| c.source_items = [i] | |
| c.bbox = item.get("bbox") | |
| if item.get("img_path"): | |
| c.images = [item["img_path"]] | |
| if c.text or c.images: | |
| chunks.append(c) | |
| continue | |
| # `image` sits alongside `chart`: different backends label the same | |
| # picture differently — what `pipeline` calls a chart, `vlm` calls an | |
| # image. Without this branch an `image` item falls through to the generic | |
| # text path, which reads only `item["text"]`, and pictures have no such | |
| # field — so BOTH its caption and its description vanish in silence. | |
| if kind in {"chart", "image"}: | |
| close() | |
| c = open_chunk(kind, page) | |
| c.text, c.generated_description = _figure_text(item, kind) | |
| c.source_items = [i] | |
| c.bbox = item.get("bbox") | |
| if item.get("img_path"): | |
| c.images = [item["img_path"]] | |
| chunks.append(c) | |
| continue | |
| # `list` keeps its content in `list_items` and has NO `text` field — the | |
| # same trap as `chart`/`image` above, and it cost more: the generic path | |
| # reads `item["text"]`, finds nothing, and drops the entire block in | |
| # silence. Measured on the BUMA standard, two dropped list blocks took | |
| # seven gold terms with them (`Uncontrollable`, `Joint survey`, | |
| # `Truck count`, `Mineplan`, `EWH`, `Fleet management`, `Controllable`) | |
| # — 0.7073 recall against 0.8537 for the parser it was meant to beat. | |
| # | |
| # List content is prose inside its section, so it joins the running text | |
| # chunk rather than opening its own. It deliberately skips heading | |
| # detection: a short bullet can look like a heading, and letting one open | |
| # a section would split a list away from the paragraph introducing it. | |
| if kind == "list": | |
| entries = [str(s).strip() for s in (item.get("list_items") or []) if str(s).strip()] | |
| if not entries: | |
| continue | |
| if running is None: | |
| running = open_chunk("text", page) | |
| running.bbox = item.get("bbox") | |
| running.source_items.append(i) | |
| running.page_idxs.append(page) | |
| pieces.append("\n".join(entries)) # verbatim, never tidied | |
| continue | |
| if kind == "equation": | |
| latex = (item.get("text") or "").strip() | |
| if running is None: | |
| running = open_chunk("text", page) | |
| running.has_formula = True | |
| running.source_items.append(i) | |
| running.page_idxs.append(page) | |
| if item.get("img_path"): | |
| running.images.append(item["img_path"]) | |
| if latex: | |
| running.latex.append(latex) # raw, for the formula branch | |
| rendered = render_latex(latex, vocabulary_for(page)) | |
| if rendered: | |
| pieces.append(rendered) # prose, so NER can find it | |
| continue | |
| # everything else: text | |
| text = (item.get("text") or "").strip() | |
| if not text: | |
| continue | |
| number, heading, level = _heading(item) | |
| if heading is not None: | |
| # BREADCRUMB: many documents reprint their heading trail at the top | |
| # of every page (BUMA repeats "2. PENJELASAN PARAMETER / | |
| # 2.1. Production Parameter" on pp. 2-8). A heading ALREADY on the | |
| # stack is the running section or one of its parents — not a new | |
| # section. Without this rule the same section splits repeatedly and | |
| # every downstream figure breaks with it. | |
| if any(text == t for _, t in stack): | |
| if running is not None: | |
| running.page_idxs.append(page) # section continues, page range grows | |
| continue | |
| close() | |
| # The heading is pushed BEFORE the chunk opens, so the chunk carries | |
| # its own heading at the end of heading_path. | |
| push_heading(level or 1, text) | |
| running = open_chunk("text", page, section_no=number, heading=heading) | |
| running.source_items = [i] | |
| running.bbox = item.get("bbox") | |
| # The heading line is NOT copied into `text` — it lives in the | |
| # `heading` field, verbatim. | |
| # | |
| # The opposite was tried, because in the BUMA standard the heading | |
| # names the term and the section body then opens "Adalah ..." without | |
| # repeating it, so the chunk defining a term does not contain that | |
| # term. But the extraction side already handles this: its span-check | |
| # haystack is assembled as `heading + text`, and its ranker treats a | |
| # heading as a mention at offset 0. | |
| # | |
| # Copying it into `text` actively hurts: the heading is counted twice | |
| # in the haystack, and its occurrence becomes a real mention, so | |
| # `mention_count` inflates — and that figure is what gets compared | |
| # against the frozen baseline (169 mentions -> 66 clusters). The v2 | |
| # comparison would shift for a reason unrelated to quality. | |
| continue | |
| if running is None: | |
| running = open_chunk("text", page) | |
| running.bbox = item.get("bbox") | |
| running.source_items.append(i) | |
| running.page_idxs.append(page) | |
| chapter = chapter_per_page.get(page) | |
| if chapter and chapter not in running.chapters: | |
| running.chapters.append(chapter) | |
| pieces.append(text) # verbatim, never tidied | |
| # Size guard: only ever relevant for documents with no detected heading. | |
| # The chunk is cut at an item boundary, so the text stays verbatim. | |
| if sum(len(x) for x in pieces) >= MAX_CHUNK_CHARS: | |
| continued_from = running | |
| close() | |
| running = open_chunk("text", page, | |
| section_no=continued_from.section_no, | |
| heading=continued_from.heading) | |
| close() | |
| return chunks | |
| def normalise_from_file( | |
| content_list: Path, | |
| doc_id: str, | |
| page_vocabulary: dict[int, frozenset[str]] | None = None, | |
| ) -> list[Chunk]: | |
| items = json.loads(content_list.read_text(encoding="utf-8")) | |
| return normalise(items, doc_id, page_vocabulary) | |