Download src/knowledge_parsing/normalize_paddle.py from DataEyond/Agentic-Service-Data-Eyond-Catalog: direct link, hf CLI and curl.
- Browser
- Download file 14 kB
-
https://huggingface.co/spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/resolve/main/src/knowledge_parsing/normalize_paddle.py
- Command line
-
hf download hf://spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/src/knowledge_parsing/normalize_paddle.py
-
curl -L -o normalize_paddle.py https://huggingface.co/spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/resolve/main/src/knowledge_parsing/normalize_paddle.py
14 kB
| """Stage 3 (PaddleOCR-VL) — turn PaddleOCR's layout blocks into per-section chunks. | |
| The PaddleOCR counterpart of `normalize_mistral.py`, added 2026-09-24. Each page | |
| result carries `prunedResult.parsing_res_list`: typed layout blocks in reading | |
| order (`block_label`, `block_content`, `block_bbox`). That is the same kind of | |
| stream Mistral's `include_blocks` gives, so the section logic here follows | |
| `normalize_mistral` and REUSES its helpers (`_heading`, markdown stripping, table | |
| and LaTeX rendering) — section boundaries are decided identically on every backend. | |
| What differs, measured on a 64-page Komatsu shop manual (HD785-7, troubleshooting | |
| part 3): | |
| - **A `figure_title` usually PRECEDES what it names** — 37 of 40 sit directly | |
| before a table ("Failure code [CA135] ..."). A title before a table becomes the | |
| table's caption: prepended to the table chunk's text (as `normalize._table_text` | |
| does for MinerU) and stored on its `Asset.caption`, so "CA135" finds its own | |
| table. A title next to an image or chart is that figure's caption and ALSO stays | |
| in the prose, because it is verbatim document text. | |
| - **Content is lightly HTML-wrapped** (`<div style="text-align: center;">…</div>`) | |
| as well as markdown (`## Title`). Both wrappers are stripped — conservatively, | |
| only known tags — and nothing else is touched. | |
| - **Figures are `<img src="imgs/…">` references**; their bytes were fetched by | |
| `paddle_ocr.run_ocr` into `markdown.images_base64` of the same page. | |
| - **Two running headers per page** (document code + chapter) — both kept as | |
| chapter context. | |
| ⚠️ Text is NEVER tidied beyond markup removal — no rejoining lines, no whitespace | |
| normalisation. See contracts.py for why. | |
| """ | |
| from __future__ import annotations | |
| import hashlib | |
| import re | |
| from collections.abc import Iterator | |
| from functools import lru_cache | |
| from pathlib import Path | |
| from typing import Any | |
| from .assets import AssetSink, store_asset, synthetic_asset_id | |
| from .contracts import Asset, Chunk | |
| from .normalize import MAX_CHUNK_CHARS, _heading | |
| from .normalize_mistral import _clean_heading, _strip_markdown | |
| from .render import render_latex, render_table | |
| # Fingerprint of the code that decides chunk content on this backend (same | |
| # mechanism as normalize_mistral.normaliser_version). The two borrowed-from | |
| # modules are included because their helpers shape this backend's text too. | |
| _CONTENT_MODULES = ("normalize_paddle.py", "normalize_mistral.py", "normalize.py", "render.py") | |
| def normaliser_version() -> str: | |
| here = Path(__file__).parent | |
| h = hashlib.sha256() | |
| for name in _CONTENT_MODULES: | |
| h.update((here / name).read_bytes()) | |
| return h.hexdigest()[:12] | |
| # Page furniture, dropped on purpose (PP-DocLayout labels). | |
| _DISCARDED = {"footer", "footer_image", "header_image", "number", "page_number", "formula_number"} | |
| _HEADER = "header" | |
| _TITLES = {"paragraph_title", "doc_title"} | |
| _CAPTIONS = {"figure_title", "table_title", "chart_title"} | |
| _FORMULAS = {"formula", "display_formula", "inline_formula"} | |
| # Figure-like blocks and the Asset.kind each becomes. | |
| _FIGURES = {"image": "figure", "chart": "chart", "seal": "figure"} | |
| # Everything a known prose label; anything else is ALSO prose (never dropped) but | |
| # reported by `unknown_labels` so a new label is noticed. | |
| _PROSE = { | |
| "text", "content", "abstract", "reference", "reference_content", "footnote", | |
| "vision_footnote", "aside_text", "algorithm", | |
| } | |
| _KNOWN = _DISCARDED | {_HEADER, "table"} | _TITLES | _CAPTIONS | _FORMULAS | set(_FIGURES) | _PROSE | |
| # Only these wrapper tags are removed — a bare `<[^>]+>` would also eat a real | |
| # "< 5 V … >" in a manual's text. | |
| _HTML_TAG = re.compile(r"</?(?:div|span|p|br|img|center|b|i|u|sup|sub|strong|em)\b[^>]*>", re.I) | |
| _IMG_SRC = re.compile(r"<img[^>]*\bsrc=\"([^\"]+)\"", re.I) | |
| def _strip_html(text: str) -> str: | |
| return _HTML_TAG.sub("", text) | |
| def _plain(raw: str) -> str: | |
| """Block content → verbatim prose: HTML wrappers and markdown markup removed.""" | |
| return _strip_markdown(_strip_html(raw)).strip() | |
| def _bbox(block: dict[str, Any]) -> list[int] | None: | |
| box = block.get("block_bbox") | |
| if isinstance(box, list) and len(box) == 4: | |
| return [int(v) for v in box] | |
| return None | |
| def _figure_key(block: dict[str, Any], page_figures: dict[str, str]) -> str | None: | |
| """The page-image key for a figure block. | |
| From the block's own `<img src>` when the job formatted block content — and | |
| otherwise from its label + bbox. The live job API returned | |
| `format_block_content: False` (2026-09-24): every image block's content was | |
| EMPTY, and all 27 figures lost their bytes, though the files themselves were | |
| fetched. PaddleOCR names each file `imgs/img_in_{label}_box_{x0}_{y0}_{x1}_{y1}.jpg` | |
| after its block — verified 28/28 on both the live result and the web-UI sample — | |
| so the key is derived, and used only if that file is actually on the page. | |
| """ | |
| match = _IMG_SRC.search(block.get("block_content") or "") | |
| if match and match.group(1).strip(): | |
| return match.group(1).strip() | |
| box = block.get("block_bbox") | |
| if isinstance(box, list) and len(box) == 4: | |
| key = f"imgs/img_in_{block.get('block_label')}_box_{'_'.join(str(int(v)) for v in box)}.jpg" | |
| if key in page_figures: | |
| return key | |
| return None | |
| def _blocks(response: dict[str, Any]) -> Iterator[tuple[int, dict[str, Any]]]: | |
| """(page_idx, block) in reading order. Page index = position in the result.""" | |
| for pidx, page in enumerate(response.get("pages") or []): | |
| for block in (page.get("prunedResult") or {}).get("parsing_res_list") or []: | |
| yield pidx, block | |
| def unknown_labels(response: dict[str, Any]) -> list[str]: | |
| """Labels this normaliser does not know — kept as prose, reported for review.""" | |
| return sorted({b.get("block_label") for _, b in _blocks(response)} - _KNOWN - {None}) | |
| def normalise_paddle( | |
| response: dict[str, Any], | |
| doc_id: str, | |
| asset_sink: AssetSink | None = None, | |
| ) -> list[Chunk]: | |
| """`paddle_ocr.run_ocr` output → chunks (section granularity, schema 0.4.0).""" | |
| pages = response.get("pages") or [] | |
| blocks = list(_blocks(response)) | |
| if not blocks: | |
| raise ValueError("no layout blocks in the PaddleOCR result") | |
| figures_b64 = { | |
| pidx: ((page.get("markdown") or {}).get("images_base64") or {}) | |
| for pidx, page in enumerate(pages) | |
| } | |
| chapters_per_page: dict[int, list[str]] = {} | |
| for pidx, b in blocks: | |
| if b.get("block_label") == _HEADER: | |
| txt = _plain(b.get("block_content") or "") | |
| if txt and txt not in chapters_per_page.setdefault(pidx, []): | |
| chapters_per_page[pidx].append(txt) | |
| def next_kind(i: int) -> str | None: | |
| """Label of the next content block on the same page (furniture skipped).""" | |
| page = blocks[i][0] | |
| for pidx, b in blocks[i + 1:]: | |
| if pidx != page: | |
| return None | |
| label = b.get("block_label") | |
| if label in _DISCARDED or label == _HEADER: | |
| continue | |
| return label | |
| return None | |
| chunks: list[Chunk] = [] | |
| running: Chunk | None = None | |
| pieces: list[str] = [] | |
| stack: list[tuple[int, str]] = [] | |
| pending_back_refs: list[tuple[str, Asset]] = [] | |
| table_caption: str | None = None # a title waiting for the next table | |
| figure_caption: str | None = None # a title waiting for the next figure | |
| uncaptioned_figure: Asset | None = None # the last figure, for a title after it | |
| def push_heading(level: int, text: str) -> None: | |
| while stack and stack[-1][0] >= level: | |
| stack.pop() | |
| stack.append((level, text)) | |
| def close() -> None: | |
| nonlocal running, pieces | |
| if running is not None: | |
| running.text = "\n\n".join(pieces).strip() | |
| if running.text or running.assets: | |
| running.page_idxs = sorted(set(running.page_idxs)) | |
| chunks.append(running) | |
| running, pieces = None, [] | |
| def open_chunk( | |
| kind: str, page: int, section_no: str | None = None, heading: str | None = None | |
| ) -> Chunk: | |
| return Chunk( | |
| chunk_id=f"{doc_id}::{len(chunks):04d}", | |
| doc_id=doc_id, kind=kind, text="", | |
| page_idx=page, page_idxs=[page], | |
| section_no=section_no, heading=heading, | |
| chapters=list(chapters_per_page.get(page, [])), | |
| heading_path=[t for _, t in stack], | |
| ) | |
| def add_prose(i: int, page: int, block: dict[str, Any], text: str) -> None: | |
| nonlocal running | |
| if running is None: | |
| running = open_chunk("text", page) | |
| running.bbox = _bbox(block) | |
| running.source_items.append(i) | |
| running.page_idxs.append(page) | |
| for chapter in chapters_per_page.get(page, []): | |
| if chapter not in running.chapters: | |
| running.chapters.append(chapter) | |
| pieces.append(text) | |
| for i, (page, block) in enumerate(blocks): | |
| label = block.get("block_label") | |
| raw = block.get("block_content") or "" | |
| if label in _DISCARDED or label == _HEADER: | |
| continue | |
| if label in _CAPTIONS: | |
| caption = _plain(raw) | |
| if not caption: | |
| continue | |
| following = next_kind(i) | |
| if following == "table": | |
| table_caption = caption # goes INTO the table chunk | |
| continue | |
| add_prose(i, page, block, caption) # verbatim evidence stays in prose | |
| if following in _FIGURES: | |
| figure_caption = caption | |
| elif uncaptioned_figure is not None: | |
| uncaptioned_figure.caption = caption | |
| uncaptioned_figure = None | |
| continue | |
| if label == "table": | |
| prose_before = running.chunk_id if running is not None else None | |
| close() | |
| html = raw.strip() | |
| c = open_chunk("table", page) | |
| body = render_table(html) | |
| c.text = f"{table_caption}\n{body}".strip() if table_caption else body | |
| c.is_tabular = True | |
| c.table_html = html or None | |
| c.source_items = [i] | |
| c.bbox = _bbox(block) | |
| asset = Asset( | |
| asset_id=synthetic_asset_id(doc_id, c.chunk_id, str(i)), | |
| kind="table", page_idx=page, bbox=_bbox(block), caption=table_caption, | |
| ) | |
| c.assets = [asset] | |
| if prose_before is not None: | |
| c.referenced_by = [prose_before] | |
| pending_back_refs.append((prose_before, asset)) | |
| if c.text: | |
| chunks.append(c) | |
| table_caption = None | |
| continue | |
| if label in _FIGURES: | |
| page_figures = figures_b64.get(page, {}) | |
| src = _figure_key(block, page_figures) | |
| identifier, storage_key = store_asset( | |
| asset_sink, src, page_figures.get(src) if src else None | |
| ) | |
| if identifier is None: # no bytes: still referenced, by coordinates | |
| identifier = synthetic_asset_id(doc_id, str(page), str(i)) | |
| asset = Asset( | |
| asset_id=identifier, kind=_FIGURES[label], storage_key=storage_key, | |
| page_idx=page, bbox=_bbox(block), caption=figure_caption, | |
| ) | |
| figure_caption = None | |
| uncaptioned_figure = asset if asset.caption is None else None | |
| if running is None: | |
| running = open_chunk("text", page) | |
| running.source_items.append(i) | |
| running.page_idxs.append(page) | |
| running.assets.append(asset) | |
| pieces.append(f"") | |
| continue | |
| if label in _FORMULAS: | |
| latex = _strip_html(raw).strip().strip("$").strip() | |
| if running is None: | |
| running = open_chunk("text", page) | |
| running.has_formula = True | |
| running.source_items.append(i) | |
| running.page_idxs.append(page) | |
| if latex: | |
| running.latex.append(latex) | |
| rendered = render_latex(latex) | |
| if rendered: | |
| pieces.append(rendered) | |
| continue | |
| # Titles and prose. A NUMBERED line opens a section (house rule, see | |
| # normalize._heading); an unnumbered title stays content. | |
| if label in _TITLES: | |
| htext, level = _clean_heading(_strip_html(raw)) | |
| else: | |
| htext, level = _plain(raw), None | |
| number, heading, hlevel = _heading({"type": "title", "text": htext, "text_level": level}) | |
| if heading is not None: | |
| if any(htext == t for _, t in stack): # reprinted breadcrumb | |
| if running is not None: | |
| running.page_idxs.append(page) | |
| continue | |
| close() | |
| push_heading(hlevel or 1, heading) | |
| running = open_chunk("text", page, section_no=number, heading=heading) | |
| running.source_items = [i] | |
| running.bbox = _bbox(block) | |
| continue | |
| text = htext if label in _TITLES else _plain(raw) | |
| if not text: | |
| continue | |
| add_prose(i, page, block, text) | |
| uncaptioned_figure = None | |
| if sum(len(x) for x in pieces) >= MAX_CHUNK_CHARS: | |
| continued_from = running | |
| close() | |
| running = open_chunk("text", page, section_no=continued_from.section_no, | |
| heading=continued_from.heading) | |
| close() | |
| by_id = {c.chunk_id: c for c in chunks} | |
| for prose_id, table_asset in pending_back_refs: | |
| host = by_id.get(prose_id) | |
| if host is not None and not any(a.asset_id == table_asset.asset_id for a in host.assets): | |
| host.assets.append(table_asset) | |
| return chunks | |