"""Stage 3 (PaddleOCR-VL) — turn PaddleOCR's layout blocks into per-section chunks. The PaddleOCR counterpart of `normalize_mistral.py`, added 2026-09-24. Each page result carries `prunedResult.parsing_res_list`: typed layout blocks in reading order (`block_label`, `block_content`, `block_bbox`). That is the same kind of stream Mistral's `include_blocks` gives, so the section logic here follows `normalize_mistral` and REUSES its helpers (`_heading`, markdown stripping, table and LaTeX rendering) — section boundaries are decided identically on every backend. What differs, measured on a 64-page Komatsu shop manual (HD785-7, troubleshooting part 3): - **A `figure_title` usually PRECEDES what it names** — 37 of 40 sit directly before a table ("Failure code [CA135] ..."). A title before a table becomes the table's caption: prepended to the table chunk's text (as `normalize._table_text` does for MinerU) and stored on its `Asset.caption`, so "CA135" finds its own table. A title next to an image or chart is that figure's caption and ALSO stays in the prose, because it is verbatim document text. - **Content is lightly HTML-wrapped** (`
…
`) as well as markdown (`## Title`). Both wrappers are stripped — conservatively, only known tags — and nothing else is touched. - **Figures are `` references**; their bytes were fetched by `paddle_ocr.run_ocr` into `markdown.images_base64` of the same page. - **Two running headers per page** (document code + chapter) — both kept as chapter context. ⚠️ Text is NEVER tidied beyond markup removal — no rejoining lines, no whitespace normalisation. See contracts.py for why. """ from __future__ import annotations import hashlib import re from collections.abc import Iterator from functools import lru_cache from pathlib import Path from typing import Any from .assets import AssetSink, store_asset, synthetic_asset_id from .contracts import Asset, Chunk from .normalize import MAX_CHUNK_CHARS, _heading from .normalize_mistral import _clean_heading, _strip_markdown from .render import render_latex, render_table # Fingerprint of the code that decides chunk content on this backend (same # mechanism as normalize_mistral.normaliser_version). The two borrowed-from # modules are included because their helpers shape this backend's text too. _CONTENT_MODULES = ("normalize_paddle.py", "normalize_mistral.py", "normalize.py", "render.py") @lru_cache(maxsize=1) def normaliser_version() -> str: here = Path(__file__).parent h = hashlib.sha256() for name in _CONTENT_MODULES: h.update((here / name).read_bytes()) return h.hexdigest()[:12] # Page furniture, dropped on purpose (PP-DocLayout labels). _DISCARDED = {"footer", "footer_image", "header_image", "number", "page_number", "formula_number"} _HEADER = "header" _TITLES = {"paragraph_title", "doc_title"} _CAPTIONS = {"figure_title", "table_title", "chart_title"} _FORMULAS = {"formula", "display_formula", "inline_formula"} # Figure-like blocks and the Asset.kind each becomes. _FIGURES = {"image": "figure", "chart": "chart", "seal": "figure"} # Everything a known prose label; anything else is ALSO prose (never dropped) but # reported by `unknown_labels` so a new label is noticed. _PROSE = { "text", "content", "abstract", "reference", "reference_content", "footnote", "vision_footnote", "aside_text", "algorithm", } _KNOWN = _DISCARDED | {_HEADER, "table"} | _TITLES | _CAPTIONS | _FORMULAS | set(_FIGURES) | _PROSE # Only these wrapper tags are removed — a bare `<[^>]+>` would also eat a real # "< 5 V … >" in a manual's text. _HTML_TAG = re.compile(r"]*>", re.I) _IMG_SRC = re.compile(r"]*\bsrc=\"([^\"]+)\"", re.I) def _strip_html(text: str) -> str: return _HTML_TAG.sub("", text) def _plain(raw: str) -> str: """Block content → verbatim prose: HTML wrappers and markdown markup removed.""" return _strip_markdown(_strip_html(raw)).strip() def _bbox(block: dict[str, Any]) -> list[int] | None: box = block.get("block_bbox") if isinstance(box, list) and len(box) == 4: return [int(v) for v in box] return None def _figure_key(block: dict[str, Any], page_figures: dict[str, str]) -> str | None: """The page-image key for a figure block. From the block's own `` when the job formatted block content — and otherwise from its label + bbox. The live job API returned `format_block_content: False` (2026-09-24): every image block's content was EMPTY, and all 27 figures lost their bytes, though the files themselves were fetched. PaddleOCR names each file `imgs/img_in_{label}_box_{x0}_{y0}_{x1}_{y1}.jpg` after its block — verified 28/28 on both the live result and the web-UI sample — so the key is derived, and used only if that file is actually on the page. """ match = _IMG_SRC.search(block.get("block_content") or "") if match and match.group(1).strip(): return match.group(1).strip() box = block.get("block_bbox") if isinstance(box, list) and len(box) == 4: key = f"imgs/img_in_{block.get('block_label')}_box_{'_'.join(str(int(v)) for v in box)}.jpg" if key in page_figures: return key return None def _blocks(response: dict[str, Any]) -> Iterator[tuple[int, dict[str, Any]]]: """(page_idx, block) in reading order. Page index = position in the result.""" for pidx, page in enumerate(response.get("pages") or []): for block in (page.get("prunedResult") or {}).get("parsing_res_list") or []: yield pidx, block def unknown_labels(response: dict[str, Any]) -> list[str]: """Labels this normaliser does not know — kept as prose, reported for review.""" return sorted({b.get("block_label") for _, b in _blocks(response)} - _KNOWN - {None}) def normalise_paddle( response: dict[str, Any], doc_id: str, asset_sink: AssetSink | None = None, ) -> list[Chunk]: """`paddle_ocr.run_ocr` output → chunks (section granularity, schema 0.4.0).""" pages = response.get("pages") or [] blocks = list(_blocks(response)) if not blocks: raise ValueError("no layout blocks in the PaddleOCR result") figures_b64 = { pidx: ((page.get("markdown") or {}).get("images_base64") or {}) for pidx, page in enumerate(pages) } chapters_per_page: dict[int, list[str]] = {} for pidx, b in blocks: if b.get("block_label") == _HEADER: txt = _plain(b.get("block_content") or "") if txt and txt not in chapters_per_page.setdefault(pidx, []): chapters_per_page[pidx].append(txt) def next_kind(i: int) -> str | None: """Label of the next content block on the same page (furniture skipped).""" page = blocks[i][0] for pidx, b in blocks[i + 1:]: if pidx != page: return None label = b.get("block_label") if label in _DISCARDED or label == _HEADER: continue return label return None chunks: list[Chunk] = [] running: Chunk | None = None pieces: list[str] = [] stack: list[tuple[int, str]] = [] pending_back_refs: list[tuple[str, Asset]] = [] table_caption: str | None = None # a title waiting for the next table figure_caption: str | None = None # a title waiting for the next figure uncaptioned_figure: Asset | None = None # the last figure, for a title after it def push_heading(level: int, text: str) -> None: while stack and stack[-1][0] >= level: stack.pop() stack.append((level, text)) def close() -> None: nonlocal running, pieces if running is not None: running.text = "\n\n".join(pieces).strip() if running.text or running.assets: running.page_idxs = sorted(set(running.page_idxs)) chunks.append(running) running, pieces = None, [] def open_chunk( kind: str, page: int, section_no: str | None = None, heading: str | None = None ) -> Chunk: return Chunk( chunk_id=f"{doc_id}::{len(chunks):04d}", doc_id=doc_id, kind=kind, text="", page_idx=page, page_idxs=[page], section_no=section_no, heading=heading, chapters=list(chapters_per_page.get(page, [])), heading_path=[t for _, t in stack], ) def add_prose(i: int, page: int, block: dict[str, Any], text: str) -> None: nonlocal running if running is None: running = open_chunk("text", page) running.bbox = _bbox(block) running.source_items.append(i) running.page_idxs.append(page) for chapter in chapters_per_page.get(page, []): if chapter not in running.chapters: running.chapters.append(chapter) pieces.append(text) for i, (page, block) in enumerate(blocks): label = block.get("block_label") raw = block.get("block_content") or "" if label in _DISCARDED or label == _HEADER: continue if label in _CAPTIONS: caption = _plain(raw) if not caption: continue following = next_kind(i) if following == "table": table_caption = caption # goes INTO the table chunk continue add_prose(i, page, block, caption) # verbatim evidence stays in prose if following in _FIGURES: figure_caption = caption elif uncaptioned_figure is not None: uncaptioned_figure.caption = caption uncaptioned_figure = None continue if label == "table": prose_before = running.chunk_id if running is not None else None close() html = raw.strip() c = open_chunk("table", page) body = render_table(html) c.text = f"{table_caption}\n{body}".strip() if table_caption else body c.is_tabular = True c.table_html = html or None c.source_items = [i] c.bbox = _bbox(block) asset = Asset( asset_id=synthetic_asset_id(doc_id, c.chunk_id, str(i)), kind="table", page_idx=page, bbox=_bbox(block), caption=table_caption, ) c.assets = [asset] if prose_before is not None: c.referenced_by = [prose_before] pending_back_refs.append((prose_before, asset)) if c.text: chunks.append(c) table_caption = None continue if label in _FIGURES: page_figures = figures_b64.get(page, {}) src = _figure_key(block, page_figures) identifier, storage_key = store_asset( asset_sink, src, page_figures.get(src) if src else None ) if identifier is None: # no bytes: still referenced, by coordinates identifier = synthetic_asset_id(doc_id, str(page), str(i)) asset = Asset( asset_id=identifier, kind=_FIGURES[label], storage_key=storage_key, page_idx=page, bbox=_bbox(block), caption=figure_caption, ) figure_caption = None uncaptioned_figure = asset if asset.caption is None else None if running is None: running = open_chunk("text", page) running.source_items.append(i) running.page_idxs.append(page) running.assets.append(asset) pieces.append(f"![](asset://{identifier})") continue if label in _FORMULAS: latex = _strip_html(raw).strip().strip("$").strip() if running is None: running = open_chunk("text", page) running.has_formula = True running.source_items.append(i) running.page_idxs.append(page) if latex: running.latex.append(latex) rendered = render_latex(latex) if rendered: pieces.append(rendered) continue # Titles and prose. A NUMBERED line opens a section (house rule, see # normalize._heading); an unnumbered title stays content. if label in _TITLES: htext, level = _clean_heading(_strip_html(raw)) else: htext, level = _plain(raw), None number, heading, hlevel = _heading({"type": "title", "text": htext, "text_level": level}) if heading is not None: if any(htext == t for _, t in stack): # reprinted breadcrumb if running is not None: running.page_idxs.append(page) continue close() push_heading(hlevel or 1, heading) running = open_chunk("text", page, section_no=number, heading=heading) running.source_items = [i] running.bbox = _bbox(block) continue text = htext if label in _TITLES else _plain(raw) if not text: continue add_prose(i, page, block, text) uncaptioned_figure = None if sum(len(x) for x in pieces) >= MAX_CHUNK_CHARS: continued_from = running close() running = open_chunk("text", page, section_no=continued_from.section_no, heading=continued_from.heading) close() by_id = {c.chunk_id: c for c in chunks} for prose_id, table_asset in pending_back_refs: host = by_id.get(prose_id) if host is not None and not any(a.asset_id == table_asset.asset_id for a in host.assets): host.assets.append(table_asset) return chunks