"""Stage 3 (PaddleOCR-VL) — turn PaddleOCR's layout blocks into per-section chunks.
The PaddleOCR counterpart of `normalize_mistral.py`, added 2026-09-24. Each page
result carries `prunedResult.parsing_res_list`: typed layout blocks in reading
order (`block_label`, `block_content`, `block_bbox`). That is the same kind of
stream Mistral's `include_blocks` gives, so the section logic here follows
`normalize_mistral` and REUSES its helpers (`_heading`, markdown stripping, table
and LaTeX rendering) — section boundaries are decided identically on every backend.
What differs, measured on a 64-page Komatsu shop manual (HD785-7, troubleshooting
part 3):
- **A `figure_title` usually PRECEDES what it names** — 37 of 40 sit directly
before a table ("Failure code [CA135] ..."). A title before a table becomes the
table's caption: prepended to the table chunk's text (as `normalize._table_text`
does for MinerU) and stored on its `Asset.caption`, so "CA135" finds its own
table. A title next to an image or chart is that figure's caption and ALSO stays
in the prose, because it is verbatim document text.
- **Content is lightly HTML-wrapped** (`
…
`)
as well as markdown (`## Title`). Both wrappers are stripped — conservatively,
only known tags — and nothing else is touched.
- **Figures are `
` references**; their bytes were fetched by
`paddle_ocr.run_ocr` into `markdown.images_base64` of the same page.
- **Two running headers per page** (document code + chapter) — both kept as
chapter context.
⚠️ Text is NEVER tidied beyond markup removal — no rejoining lines, no whitespace
normalisation. See contracts.py for why.
"""
from __future__ import annotations
import hashlib
import re
from collections.abc import Iterator
from functools import lru_cache
from pathlib import Path
from typing import Any
from .assets import AssetSink, store_asset, synthetic_asset_id
from .contracts import Asset, Chunk
from .normalize import MAX_CHUNK_CHARS, _heading
from .normalize_mistral import _clean_heading, _strip_markdown
from .render import render_latex, render_table
# Fingerprint of the code that decides chunk content on this backend (same
# mechanism as normalize_mistral.normaliser_version). The two borrowed-from
# modules are included because their helpers shape this backend's text too.
_CONTENT_MODULES = ("normalize_paddle.py", "normalize_mistral.py", "normalize.py", "render.py")
@lru_cache(maxsize=1)
def normaliser_version() -> str:
here = Path(__file__).parent
h = hashlib.sha256()
for name in _CONTENT_MODULES:
h.update((here / name).read_bytes())
return h.hexdigest()[:12]
# Page furniture, dropped on purpose (PP-DocLayout labels).
_DISCARDED = {"footer", "footer_image", "header_image", "number", "page_number", "formula_number"}
_HEADER = "header"
_TITLES = {"paragraph_title", "doc_title"}
_CAPTIONS = {"figure_title", "table_title", "chart_title"}
_FORMULAS = {"formula", "display_formula", "inline_formula"}
# Figure-like blocks and the Asset.kind each becomes.
_FIGURES = {"image": "figure", "chart": "chart", "seal": "figure"}
# Everything a known prose label; anything else is ALSO prose (never dropped) but
# reported by `unknown_labels` so a new label is noticed.
_PROSE = {
"text", "content", "abstract", "reference", "reference_content", "footnote",
"vision_footnote", "aside_text", "algorithm",
}
_KNOWN = _DISCARDED | {_HEADER, "table"} | _TITLES | _CAPTIONS | _FORMULAS | set(_FIGURES) | _PROSE
# Only these wrapper tags are removed — a bare `<[^>]+>` would also eat a real
# "< 5 V … >" in a manual's text.
_HTML_TAG = re.compile(r"?(?:div|span|p|br|img|center|b|i|u|sup|sub|strong|em)\b[^>]*>", re.I)
_IMG_SRC = re.compile(r"
]*\bsrc=\"([^\"]+)\"", re.I)
def _strip_html(text: str) -> str:
return _HTML_TAG.sub("", text)
def _plain(raw: str) -> str:
"""Block content → verbatim prose: HTML wrappers and markdown markup removed."""
return _strip_markdown(_strip_html(raw)).strip()
def _bbox(block: dict[str, Any]) -> list[int] | None:
box = block.get("block_bbox")
if isinstance(box, list) and len(box) == 4:
return [int(v) for v in box]
return None
def _figure_key(block: dict[str, Any], page_figures: dict[str, str]) -> str | None:
"""The page-image key for a figure block.
From the block's own `
` when the job formatted block content — and
otherwise from its label + bbox. The live job API returned
`format_block_content: False` (2026-09-24): every image block's content was
EMPTY, and all 27 figures lost their bytes, though the files themselves were
fetched. PaddleOCR names each file `imgs/img_in_{label}_box_{x0}_{y0}_{x1}_{y1}.jpg`
after its block — verified 28/28 on both the live result and the web-UI sample —
so the key is derived, and used only if that file is actually on the page.
"""
match = _IMG_SRC.search(block.get("block_content") or "")
if match and match.group(1).strip():
return match.group(1).strip()
box = block.get("block_bbox")
if isinstance(box, list) and len(box) == 4:
key = f"imgs/img_in_{block.get('block_label')}_box_{'_'.join(str(int(v)) for v in box)}.jpg"
if key in page_figures:
return key
return None
def _blocks(response: dict[str, Any]) -> Iterator[tuple[int, dict[str, Any]]]:
"""(page_idx, block) in reading order. Page index = position in the result."""
for pidx, page in enumerate(response.get("pages") or []):
for block in (page.get("prunedResult") or {}).get("parsing_res_list") or []:
yield pidx, block
def unknown_labels(response: dict[str, Any]) -> list[str]:
"""Labels this normaliser does not know — kept as prose, reported for review."""
return sorted({b.get("block_label") for _, b in _blocks(response)} - _KNOWN - {None})
def normalise_paddle(
response: dict[str, Any],
doc_id: str,
asset_sink: AssetSink | None = None,
) -> list[Chunk]:
"""`paddle_ocr.run_ocr` output → chunks (section granularity, schema 0.4.0)."""
pages = response.get("pages") or []
blocks = list(_blocks(response))
if not blocks:
raise ValueError("no layout blocks in the PaddleOCR result")
figures_b64 = {
pidx: ((page.get("markdown") or {}).get("images_base64") or {})
for pidx, page in enumerate(pages)
}
chapters_per_page: dict[int, list[str]] = {}
for pidx, b in blocks:
if b.get("block_label") == _HEADER:
txt = _plain(b.get("block_content") or "")
if txt and txt not in chapters_per_page.setdefault(pidx, []):
chapters_per_page[pidx].append(txt)
def next_kind(i: int) -> str | None:
"""Label of the next content block on the same page (furniture skipped)."""
page = blocks[i][0]
for pidx, b in blocks[i + 1:]:
if pidx != page:
return None
label = b.get("block_label")
if label in _DISCARDED or label == _HEADER:
continue
return label
return None
chunks: list[Chunk] = []
running: Chunk | None = None
pieces: list[str] = []
stack: list[tuple[int, str]] = []
pending_back_refs: list[tuple[str, Asset]] = []
table_caption: str | None = None # a title waiting for the next table
figure_caption: str | None = None # a title waiting for the next figure
uncaptioned_figure: Asset | None = None # the last figure, for a title after it
def push_heading(level: int, text: str) -> None:
while stack and stack[-1][0] >= level:
stack.pop()
stack.append((level, text))
def close() -> None:
nonlocal running, pieces
if running is not None:
running.text = "\n\n".join(pieces).strip()
if running.text or running.assets:
running.page_idxs = sorted(set(running.page_idxs))
chunks.append(running)
running, pieces = None, []
def open_chunk(
kind: str, page: int, section_no: str | None = None, heading: str | None = None
) -> Chunk:
return Chunk(
chunk_id=f"{doc_id}::{len(chunks):04d}",
doc_id=doc_id, kind=kind, text="",
page_idx=page, page_idxs=[page],
section_no=section_no, heading=heading,
chapters=list(chapters_per_page.get(page, [])),
heading_path=[t for _, t in stack],
)
def add_prose(i: int, page: int, block: dict[str, Any], text: str) -> None:
nonlocal running
if running is None:
running = open_chunk("text", page)
running.bbox = _bbox(block)
running.source_items.append(i)
running.page_idxs.append(page)
for chapter in chapters_per_page.get(page, []):
if chapter not in running.chapters:
running.chapters.append(chapter)
pieces.append(text)
for i, (page, block) in enumerate(blocks):
label = block.get("block_label")
raw = block.get("block_content") or ""
if label in _DISCARDED or label == _HEADER:
continue
if label in _CAPTIONS:
caption = _plain(raw)
if not caption:
continue
following = next_kind(i)
if following == "table":
table_caption = caption # goes INTO the table chunk
continue
add_prose(i, page, block, caption) # verbatim evidence stays in prose
if following in _FIGURES:
figure_caption = caption
elif uncaptioned_figure is not None:
uncaptioned_figure.caption = caption
uncaptioned_figure = None
continue
if label == "table":
prose_before = running.chunk_id if running is not None else None
close()
html = raw.strip()
c = open_chunk("table", page)
body = render_table(html)
c.text = f"{table_caption}\n{body}".strip() if table_caption else body
c.is_tabular = True
c.table_html = html or None
c.source_items = [i]
c.bbox = _bbox(block)
asset = Asset(
asset_id=synthetic_asset_id(doc_id, c.chunk_id, str(i)),
kind="table", page_idx=page, bbox=_bbox(block), caption=table_caption,
)
c.assets = [asset]
if prose_before is not None:
c.referenced_by = [prose_before]
pending_back_refs.append((prose_before, asset))
if c.text:
chunks.append(c)
table_caption = None
continue
if label in _FIGURES:
page_figures = figures_b64.get(page, {})
src = _figure_key(block, page_figures)
identifier, storage_key = store_asset(
asset_sink, src, page_figures.get(src) if src else None
)
if identifier is None: # no bytes: still referenced, by coordinates
identifier = synthetic_asset_id(doc_id, str(page), str(i))
asset = Asset(
asset_id=identifier, kind=_FIGURES[label], storage_key=storage_key,
page_idx=page, bbox=_bbox(block), caption=figure_caption,
)
figure_caption = None
uncaptioned_figure = asset if asset.caption is None else None
if running is None:
running = open_chunk("text", page)
running.source_items.append(i)
running.page_idxs.append(page)
running.assets.append(asset)
pieces.append(f"")
continue
if label in _FORMULAS:
latex = _strip_html(raw).strip().strip("$").strip()
if running is None:
running = open_chunk("text", page)
running.has_formula = True
running.source_items.append(i)
running.page_idxs.append(page)
if latex:
running.latex.append(latex)
rendered = render_latex(latex)
if rendered:
pieces.append(rendered)
continue
# Titles and prose. A NUMBERED line opens a section (house rule, see
# normalize._heading); an unnumbered title stays content.
if label in _TITLES:
htext, level = _clean_heading(_strip_html(raw))
else:
htext, level = _plain(raw), None
number, heading, hlevel = _heading({"type": "title", "text": htext, "text_level": level})
if heading is not None:
if any(htext == t for _, t in stack): # reprinted breadcrumb
if running is not None:
running.page_idxs.append(page)
continue
close()
push_heading(hlevel or 1, heading)
running = open_chunk("text", page, section_no=number, heading=heading)
running.source_items = [i]
running.bbox = _bbox(block)
continue
text = htext if label in _TITLES else _plain(raw)
if not text:
continue
add_prose(i, page, block, text)
uncaptioned_figure = None
if sum(len(x) for x in pieces) >= MAX_CHUNK_CHARS:
continued_from = running
close()
running = open_chunk("text", page, section_no=continued_from.section_no,
heading=continued_from.heading)
close()
by_id = {c.chunk_id: c for c in chunks}
for prose_id, table_asset in pending_back_refs:
host = by_id.get(prose_id)
if host is not None and not any(a.asset_id == table_asset.asset_id for a in host.assets):
host.assets.append(table_asset)
return chunks