"""Stage 3 (Mistral) — turn Mistral OCR's block stream into per-section chunks. The counterpart of `normalize.py`, for the Mistral OCR backend. With `include_blocks=True` Mistral returns a per-page list of typed blocks (`title` / `text` / `equation` / `table` / `caption` / `list` / `image` / `header` / `footer`), each with a bbox and its content — a structure ~1:1 with MinerU's `content_list.json`. So this is a close port of `normalize.py`, and it REUSES that module's `_heading` logic verbatim so section boundaries are decided identically on both backends. Two things it does that `normalize.py` does not have to: - **Strip markdown markup from text.** Mistral renders content as markdown, so a text line arrives as `*Production* yang **digunakan** ...`. Those `*`/`#` are markup Mistral added, NOT characters in the source PDF — and the extraction span-check locates LLM-quoted spans LITERALLY inside `text`, so leaving them in makes a quote of "Production" fail to match. Stripping markup is the same move `render.py` makes for LaTeX/HTML; it is NOT reflowing. Whitespace and line breaks are still left exactly as they are. - **Correlate image descriptions across the response.** A block of type `image` carries only `![img-N](...)` + an `image_id`; the model-written description lives at the page level in `pages[].images[].image_annotation` (produced by `bbox_annotation`). They are matched by id here. ⚠️ Text is NEVER tidied beyond markup stripping — no rejoining lines, no whitespace normalisation. See contracts.py for why. """ from __future__ import annotations import hashlib import json import re from collections.abc import Iterator from functools import lru_cache from pathlib import Path from typing import Any from .assets import AssetSink, store_asset, synthetic_asset_id from .contracts import Asset, Chunk from .normalize import MAX_CHUNK_CHARS, _heading # identical section-boundary logic from .render import render_latex, render_table # Fingerprint of the code that decides chunk CONTENT on this backend — same # mechanism and reasoning as normalize.normaliser_version (see there). Covers # this module and render.py; contracts.py is deliberately excluded (its shape is # versioned by schema_version). _CONTENT_MODULES = ("normalize_mistral.py", "render.py") @lru_cache(maxsize=1) def normaliser_version() -> str: here = Path(__file__).parent h = hashlib.sha256() for name in _CONTENT_MODULES: h.update((here / name).read_bytes()) return h.hexdigest()[:12] # Page furniture, dropped on purpose — matches normalize._DISCARDED. `header` # is handled separately (it becomes chapter context). `signature` is Mistral's # label for the cover-page "Disusun/Disetujui" stamps, whose content is empty in # the source anyway. _DISCARDED = {"page_number", "footer"} # Recorded on every Asset so a later swap of the description model is traceable # rather than silently changing what the field means. _DESCRIPTION_SOURCE = "mistral-ocr/bbox_annotation" _LEADING_HASHES = re.compile(r"^\s*(#{1,6})\s*") _MD_IMAGE = re.compile(r"!\[[^\]]*\]\([^)]*\)") # Same shape, but capturing the target so a block's own markdown can be read # for the image id it refers to (see _block_image_id). _MD_IMAGE_TARGET = re.compile(r"!\[[^\]]*\]\(([^)]*)\)") _MD_LINK = re.compile(r"\[([^\]]*)\]\([^)]*\)") _MD_BOLD = re.compile(r"\*\*(.+?)\*\*", re.S) _MD_ITALIC = re.compile(r"\*(.+?)\*", re.S) # Inline math inside prose: "... nilai $x_i$ ...". Rendered to readable text for # the same reason render.render_table renders math inside table cells. _INLINE_MATH = re.compile(r"\$([^$]{1,400})\$") def _strip_markdown(text: str, vocabulary: frozenset[str] | None = None) -> str: """Remove markup Mistral ADDED, keep the document's own characters. Deliberately conservative: only paired `*`/`**` emphasis, heading `#`, backticks, image and link syntax, and inline `$…$`. Underscores are left untouched — they appear in real variable names (`INPR_Hours`) far more often than as emphasis, and stripping them would damage content. Whitespace and newlines are preserved: this is markup removal, not reflow. """ text = _MD_IMAGE.sub("", text) # images: drop text = _MD_LINK.sub(r"\1", text) # links: keep label text = _INLINE_MATH.sub(lambda m: " " + render_latex(m.group(1), vocabulary) + " ", text) text = _LEADING_HASHES.sub("", text) text = _MD_BOLD.sub(r"\1", text) text = _MD_ITALIC.sub(r"\1", text) return text.replace("`", "") def _clean_heading(raw: str) -> tuple[str, int | None]: """A `title` block's markdown → (plain heading text, level from '#' depth).""" m = _LEADING_HASHES.match(raw) level = len(m.group(1)) if m else None text = _LEADING_HASHES.sub("", raw) text = _MD_BOLD.sub(r"\1", text) text = _MD_ITALIC.sub(r"\1", text) return text.strip(), level def _bbox(block: dict[str, Any]) -> list[int] | None: keys = ("top_left_x", "top_left_y", "bottom_right_x", "bottom_right_y") if all(k in block and block[k] is not None for k in keys): return [int(block[k]) for k in keys] return None def _blocks_with_pages(response: dict[str, Any]) -> Iterator[tuple[int, dict[str, Any]]]: """Flatten pages[].blocks into (page_idx, block), page order preserved.""" for page in response.get("pages", []): pidx = int(page.get("index", 0)) for block in page.get("blocks") or []: yield pidx, block def _image_descriptions(response: dict[str, Any]) -> dict[str, str]: """image_id -> description, from the page-level annotations (bbox_annotation). Empty when the OCR call did not request annotations; description then stays None, exactly as it does on MinerU's `pipeline` backend. """ out: dict[str, str] = {} for page in response.get("pages", []): for im in page.get("images") or []: ann = im.get("image_annotation") iid = im.get("id") if iid and ann: out[iid] = _unwrap_description(ann) return out def _unwrap_description(annotation: Any) -> str: """Pull the prose out of an annotation payload. Mistral returns bbox annotations JSON-WRAPPED — literally `{"description": "The diagram illustrates…"}` — and 0.3.x stored that string raw, braces and all, so every consumer got JSON where it expected a sentence. Unwrapped here, at the one place that reads the field. Never raises: anything unparseable falls back to the string form, because a slightly ugly description is worth more than a failed parse. """ if isinstance(annotation, dict): value = annotation.get("description") or annotation.get("text") return str(value) if value else json.dumps(annotation, ensure_ascii=False) if isinstance(annotation, str): text = annotation.strip() if text.startswith("{"): try: parsed = json.loads(text) except (ValueError, TypeError): return text if isinstance(parsed, dict): value = parsed.get("description") or parsed.get("text") if value: return str(value) return text return text return str(annotation) def _block_image_id(block: dict[str, Any]) -> str | None: """The image this block refers to, preferring the id in its own markdown. ⚠️ **`block["image_id"]` is not trustworthy.** Measured on the Open Pit textbook: page 6 carries two image blocks, and the FIRST one's `content` reads `![img-1.jpeg](img-1.jpeg)` while its `image_id` field says **`img-2.jpeg`** — so both blocks claim the same image. Keying on `image_id` therefore resolved both to img-2's bytes, gave two different figures one content hash, and lost img-1 entirely (7 files written for 8 figures). The `content` markdown is the document's own reference and was correct in every case, so it wins; `image_id` is the fallback for a block whose content carries no reference. This also corrects an earlier diagnosis: the ids are globally unique across the response (`img-0` … `img-7`), so the duplication was never page-local naming — it was this mislabelled field. """ match = _MD_IMAGE_TARGET.search(block.get("content") or "") if match and match.group(1).strip(): return match.group(1).strip() return block.get("image_id") def _image_payloads(response: dict[str, Any]) -> dict[str, str]: """image_id -> base64 payload. Empty when `include_image_base64` was off. 0.3.x never read these: the bytes were returned by the API, recorded nowhere, and `Chunk.images` pointed at filenames that did not exist on disk. 0.4.0 content-addresses and persists them. """ out: dict[str, str] = {} for page in response.get("pages", []): for im in page.get("images") or []: iid, payload = im.get("id"), im.get("image_base64") if iid and payload: out[iid] = payload return out def _chunk_by_page( blocks: list[tuple[int, dict[str, Any]]], descriptions: dict[str, str], chapter_per_page: dict[int, str], doc_id: str, ) -> list[Chunk]: """Page granularity: all prose on a page → one chunk; tables/images their own. Recall-only. A page chunk has no single heading, so `heading_path` / `section_no` stay empty and the extraction outline degrades — section granularity (the default) is the real path. """ def chapters(page: int) -> list[str]: return [chapter_per_page[page]] if page in chapter_per_page else [] chunks: list[Chunk] = [] by_page: dict[int, list[tuple[int, dict[str, Any]]]] = {} for i, (pidx, block) in enumerate(blocks): by_page.setdefault(pidx, []).append((i, block)) for pidx in sorted(by_page): pieces: list[str] = [] latex: list[str] = [] src: list[int] = [] has_formula = False prose_bbox: list[int] | None = None for i, block in by_page[pidx]: kind = block.get("type") if kind in _DISCARDED or kind == "header": continue if kind == "table": c = Chunk( chunk_id=f"{doc_id}::{len(chunks):04d}", doc_id=doc_id, kind="table", text=render_table(block.get("content") or ""), page_idx=pidx, page_idxs=[pidx], chapters=chapters(pidx), is_tabular=True, table_html=block.get("content") or None, source_items=[i], bbox=_bbox(block), ) if c.text or c.images: chunks.append(c) continue if kind == "image": iid = block.get("image_id") chunks.append(Chunk( chunk_id=f"{doc_id}::{len(chunks):04d}", doc_id=doc_id, kind="image", text="", page_idx=pidx, page_idxs=[pidx], chapters=chapters(pidx), generated_description=descriptions.get(iid) if iid else None, source_items=[i], bbox=_bbox(block), images=[iid] if iid else [], )) continue if kind == "equation": raw = (block.get("content") or "").strip().strip("$").strip() has_formula = True src.append(i) if raw: latex.append(raw) rendered = render_latex(raw) if rendered: pieces.append(rendered) continue # text / title / list / caption / signature -> prose. A numbered # heading is NOT a boundary in page mode; it stays as prose text. text = _strip_markdown(block.get("content") or "").rstrip() if not text.strip(): continue if prose_bbox is None: prose_bbox = _bbox(block) src.append(i) pieces.append(text) if pieces: chunks.append(Chunk( chunk_id=f"{doc_id}::{len(chunks):04d}", doc_id=doc_id, kind="equation" if has_formula else "text", text="\n\n".join(pieces).strip(), page_idx=pidx, page_idxs=[pidx], chapters=chapters(pidx), has_formula=has_formula, latex=latex, source_items=src, bbox=prose_bbox, )) return chunks def normalise_mistral( response: dict[str, Any], doc_id: str, granularity: str = "section", asset_sink: AssetSink | None = None, ) -> list[Chunk]: """Mistral OCR response (include_blocks=True) → chunks. `granularity` decides how prose is grouped, and only that: - "section" (default): a numbered heading opens a new chunk, mirroring normalize.py. This is the default because the extraction half derives the document `outline` from `heading_path` and drives `source_wording` from `heading` — page grouping leaves both EMPTY (a page chunk has no single heading), which silently degrades a locked product behaviour. - "page": all prose on a page becomes ONE chunk. Higher raw term recall (bigger context per NER window) but no per-chunk heading, so only fit for a recall-only measurement, never for the real pipeline. The section/page recall gap (0.878 vs 0.951 on the BUMA fixture) is the NER filter's, not content: every gold term is present in the text either way, and closing it is the extraction side's lever (span-filter overlap resolution). In BOTH modes tables and images are their OWN chunks, so `table_html` stays one-table-per-chunk regardless. Switching modes is a one-word change and never touches the extraction half — the artifact shape is identical. No `page_vocabulary` parameter: Mistral does not spell formulas out one character at a time, so the word-boundary recovery that MinerU needs has nothing to do here. """ if granularity not in {"page", "section"}: raise ValueError(f"granularity must be 'page' or 'section', got {granularity!r}") blocks = list(_blocks_with_pages(response)) if not blocks: raise ValueError( "no blocks in Mistral response — call ocr.process with include_blocks=True" ) descriptions = _image_descriptions(response) # Chapter context per page, from the running header (type 'header'). chapter_per_page: dict[int, str] = {} for pidx, b in blocks: if b.get("type") == "header": txt = _strip_markdown(b.get("content") or "").strip() if txt: chapter_per_page.setdefault(pidx, txt) # --- page mode: recall-only; a page chunk has no single heading --- if granularity == "page": # ⚠️ FENCED OFF AT 0.4.0, deliberately, rather than given the asset # treatment. Page mode emits 0.3.x-shaped chunks — images standalone with # empty text, no `assets` — so letting it run would quietly produce an # artifact that CLAIMS schema_version 0.4.0 while having the old shape. # Two normalisers silently diverging is exactly how the `type: "list"` # content drop survived unnoticed and cost five gold terms. # # Page mode exists only for a recall-only measurement (it nulls # `heading_path`, which the outline and `source_wording` both need), so it # is not worth duplicating the asset logic into. Restore by porting the # image/table branches from the section path below. raise NotImplementedError( "granularity='page' is not supported at schema 0.4.0: it would emit " "0.3.x-shaped chunks (standalone image chunks, no assets) under a " "0.4.0 version stamp. Use granularity='section', or port the asset " "handling into _chunk_by_page first." ) # --- section mode (default): a numbered heading opens a new chunk (mirrors normalize.py) --- chunks: list[Chunk] = [] running: Chunk | None = None pieces: list[str] = [] stack: list[tuple[int, str]] = [] payloads = _image_payloads(response) # A table's asset has to be added to the PROSE chunk that precedes it, but # that chunk is already closed and appended by then — so the back-reference is # queued and resolved once, after the loop. _pending_back_refs: list[tuple[str, Asset]] = [] # A printed caption arrives as its own `caption` block AFTER its image block. # The figure waiting for one is held here; the caption is attached to the asset # AND left in the prose text, because it is verbatim document content. _pending_caption_for: list[Asset] = [] def push_heading(level: int, text: str) -> None: while stack and stack[-1][0] >= level: stack.pop() stack.append((level, text)) def close() -> None: nonlocal running, pieces if running is not None: running.text = "\n\n".join(pieces).strip() if running.text or running.assets or running.images: running.page_idxs = sorted(set(running.page_idxs)) chunks.append(running) running, pieces = None, [] def open_chunk( kind: str, page: int, section_no: str | None = None, heading: str | None = None ) -> Chunk: return Chunk( chunk_id=f"{doc_id}::{len(chunks):04d}", doc_id=doc_id, kind=kind, text="", page_idx=page, page_idxs=[page], section_no=section_no, heading=heading, chapters=[chapter_per_page[page]] if page in chapter_per_page else [], heading_path=[t for _, t in stack], ) for i, (page, block) in enumerate(blocks): kind = block.get("type") if kind in _DISCARDED or kind == "header": continue if kind == "table": # Tables stay their OWN chunk (0.4.0 kept this deliberately): unlike a # figure, a table's linearised text carries real terms the filter # finds — 74.6% of the Komatsu manual's top-3 evidence is tabular. What # 0.4.0 adds is reachability: an `Asset` plus a two-way pointer, so the # prose that introduces a table can be followed to it and back. prose_before = running.chunk_id if running is not None else None close() c = open_chunk("table", page) c.text = render_table(block.get("content") or "") c.is_tabular = True c.table_html = block.get("content") or None c.source_items = [i] c.bbox = _bbox(block) table_asset = Asset( # No bytes to hash, so the id is derived from stable document # coordinates instead — deterministic across re-parses, which is # what a durable pointer needs. asset_id=synthetic_asset_id(doc_id, c.chunk_id, str(i)), kind="table", page_idx=page, bbox=_bbox(block), ) c.assets = [table_asset] if prose_before is not None: c.referenced_by = [prose_before] _pending_back_refs.append((prose_before, table_asset)) if c.text or c.assets: chunks.append(c) continue if kind == "image": # ── 0.4.0: a figure is NO LONGER ITS OWN CHUNK. ──────────────────── # It used to be, with an EMPTY `text`, which made it unreachable by # construction: no text -> no term mentions -> joins no cluster -> # never selected as evidence. Measured across three documents: 13 # figures, every one described by a vision model we pay for, and ZERO # appearing in any cluster's evidence. Meanwhile the prose saying # "Figure 1.1" had its `![](…)` reference stripped, so nothing # connected the sentence to the picture. # # Now the reference is folded into the RUNNING prose chunk at its # reading-order position — exactly where the document put it — and the # figure travels as an `Asset` on that chunk. iid = _block_image_id(block) identifier, storage_key = store_asset( asset_sink, iid, payloads.get(iid) if iid else None ) # Fall back to document coordinates when the parser returned no bytes, # so a figure is still referenced rather than silently vanishing. if identifier is None: identifier = synthetic_asset_id(doc_id, str(page), str(i)) asset = Asset( asset_id=identifier, kind="figure", storage_key=storage_key, page_idx=page, bbox=_bbox(block), description=descriptions.get(iid) if iid else None, description_source=_DESCRIPTION_SOURCE if iid else None, ) # P5 edge case: a figure can precede any prose on its page (Open Pit # p.3 is close to this shape). Open a chunk rather than drop the asset. if running is None: running = open_chunk("text", page) running.source_items.append(i) running.page_idxs.append(page) running.assets.append(asset) # Deprecated alias, one version only. if iid: running.images.append(iid) # The reference itself, inline in the prose. `asset://` rather than the # parser's filename because the filename is page-local and collides. pieces.append(f"![](asset://{identifier})") _pending_caption_for.append(asset) continue if kind == "equation": latex = (block.get("content") or "").strip().strip("$").strip() if running is None: running = open_chunk("text", page) running.has_formula = True running.source_items.append(i) running.page_idxs.append(page) if latex: running.latex.append(latex) # raw, for the formula branch rendered = render_latex(latex) if rendered: pieces.append(rendered) # prose, so NER can find it continue raw = block.get("content") or "" # `list` and `caption` are ALWAYS content, never a section boundary: a # bullet or a caption can look like a numbered heading, and letting one # open a section would split a list from the paragraph introducing it # (see normalize.py's list note). if kind in {"list", "caption"}: text = _strip_markdown(raw).rstrip() if text.strip(): if running is None: running = open_chunk("text", page) running.bbox = _bbox(block) running.source_items.append(i) running.page_idxs.append(page) pieces.append(text) # A caption directly after a figure names that figure. Attach it to # the asset so a consumer can render "Figure 1.1 …" beside the # image — and LEAVE it in `pieces` too, because it is printed in # the document: verbatim, quotable as evidence, span-checkable. # Copying it is the point; moving it would silently delete # evidence that survived 17/17 times on the Open Pit textbook. if kind == "caption" and _pending_caption_for: _pending_caption_for.pop(0).caption = text.strip() continue # `text` / `title` / `signature`: a NUMBERED line may open a section. # normalize._heading is applied to BOTH text and title items on MinerU — # here too, because Mistral labels many numbered headings as `text`, not # `title`. Checking only `title` left whole sections unopened and every # equation piled into one chunk. A title's level comes from its `#` # depth; a text line's from numbering depth (fallback inside _heading). if kind == "title": htext, level = _clean_heading(raw) else: htext, level = _strip_markdown(raw).strip(), None number, heading, hlevel = _heading( {"type": "title", "text": htext, "text_level": level} ) if heading is not None: if any(htext == t for _, t in stack): # reprinted breadcrumb if running is not None: running.page_idxs.append(page) continue close() push_heading(hlevel or 1, heading) running = open_chunk("text", page, section_no=number, heading=heading) running.source_items = [i] running.bbox = _bbox(block) # heading line NOT copied into text — lives in `heading`. See # normalize.py for the measured reason (mention_count inflation). continue text = _strip_markdown(raw).rstrip() if not text.strip(): continue if running is None: running = open_chunk("text", page) running.bbox = _bbox(block) running.source_items.append(i) running.page_idxs.append(page) chapter = chapter_per_page.get(page) if chapter and chapter not in running.chapters: running.chapters.append(chapter) pieces.append(text) # verbatim (markup only removed) if sum(len(x) for x in pieces) >= MAX_CHUNK_CHARS: continued_from = running close() running = open_chunk("text", page, section_no=continued_from.section_no, heading=continued_from.heading) close() # Two-way table pointers, resolved once the loop is done. The table chunk # already names the prose chunk in `referenced_by`; this adds the table's asset # to that prose chunk so the link is navigable from the prose side too — the # direction a reader, and a model reading evidence, actually travels. Deferred # rather than done inline because the prose chunk is already closed and # appended by the time its table is reached. by_id = {c.chunk_id: c for c in chunks} for prose_id, table_asset in _pending_back_refs: host = by_id.get(prose_id) if host is not None and not any( a.asset_id == table_asset.asset_id for a in host.assets ): host.assets.append(table_asset) return chunks