ishaq101's picture
/fix parsing and term extract (#21)
f07443e
Raw History Blame Contribute Delete
27.1 kB
"""Stage 3 (Mistral) — turn Mistral OCR's block stream into per-section chunks.
The counterpart of `normalize.py`, for the Mistral OCR backend. With
`include_blocks=True` Mistral returns a per-page list of typed blocks
(`title` / `text` / `equation` / `table` / `caption` / `list` / `image` /
`header` / `footer`), each with a bbox and its content — a structure ~1:1 with
MinerU's `content_list.json`. So this is a close port of `normalize.py`, and it
REUSES that module's `_heading` logic verbatim so section boundaries are decided
identically on both backends.
Two things it does that `normalize.py` does not have to:
- **Strip markdown markup from text.** Mistral renders content as markdown, so
a text line arrives as `*Production* yang **digunakan** ...`. Those `*`/`#`
are markup Mistral added, NOT characters in the source PDF — and the
extraction span-check locates LLM-quoted spans LITERALLY inside `text`, so
leaving them in makes a quote of "Production" fail to match. Stripping markup
is the same move `render.py` makes for LaTeX/HTML; it is NOT reflowing.
Whitespace and line breaks are still left exactly as they are.
- **Correlate image descriptions across the response.** A block of type
`image` carries only `![img-N](...)` + an `image_id`; the model-written
description lives at the page level in `pages[].images[].image_annotation`
(produced by `bbox_annotation`). They are matched by id here.
⚠️ Text is NEVER tidied beyond markup stripping — no rejoining lines, no
whitespace normalisation. See contracts.py for why.
"""
from __future__ import annotations
import hashlib
import json
import re
from collections.abc import Iterator
from functools import lru_cache
from pathlib import Path
from typing import Any
from .assets import AssetSink, store_asset, synthetic_asset_id
from .contracts import Asset, Chunk
from .normalize import MAX_CHUNK_CHARS, _heading # identical section-boundary logic
from .render import render_latex, render_table
# Fingerprint of the code that decides chunk CONTENT on this backend — same
# mechanism and reasoning as normalize.normaliser_version (see there). Covers
# this module and render.py; contracts.py is deliberately excluded (its shape is
# versioned by schema_version).
_CONTENT_MODULES = ("normalize_mistral.py", "render.py")
@lru_cache(maxsize=1)
def normaliser_version() -> str:
here = Path(__file__).parent
h = hashlib.sha256()
for name in _CONTENT_MODULES:
h.update((here / name).read_bytes())
return h.hexdigest()[:12]
# Page furniture, dropped on purpose — matches normalize._DISCARDED. `header`
# is handled separately (it becomes chapter context). `signature` is Mistral's
# label for the cover-page "Disusun/Disetujui" stamps, whose content is empty in
# the source anyway.
_DISCARDED = {"page_number", "footer"}
# Recorded on every Asset so a later swap of the description model is traceable
# rather than silently changing what the field means.
_DESCRIPTION_SOURCE = "mistral-ocr/bbox_annotation"
_LEADING_HASHES = re.compile(r"^\s*(#{1,6})\s*")
_MD_IMAGE = re.compile(r"!\[[^\]]*\]\([^)]*\)")
# Same shape, but capturing the target so a block's own markdown can be read
# for the image id it refers to (see _block_image_id).
_MD_IMAGE_TARGET = re.compile(r"!\[[^\]]*\]\(([^)]*)\)")
_MD_LINK = re.compile(r"\[([^\]]*)\]\([^)]*\)")
_MD_BOLD = re.compile(r"\*\*(.+?)\*\*", re.S)
_MD_ITALIC = re.compile(r"\*(.+?)\*", re.S)
# Inline math inside prose: "... nilai $x_i$ ...". Rendered to readable text for
# the same reason render.render_table renders math inside table cells.
_INLINE_MATH = re.compile(r"\$([^$]{1,400})\$")
def _strip_markdown(text: str, vocabulary: frozenset[str] | None = None) -> str:
"""Remove markup Mistral ADDED, keep the document's own characters.
Deliberately conservative: only paired `*`/`**` emphasis, heading `#`,
backticks, image and link syntax, and inline `$…$`. Underscores are left
untouched — they appear in real variable names (`INPR_Hours`) far more often
than as emphasis, and stripping them would damage content. Whitespace and
newlines are preserved: this is markup removal, not reflow.
"""
text = _MD_IMAGE.sub("", text) # images: drop
text = _MD_LINK.sub(r"\1", text) # links: keep label
text = _INLINE_MATH.sub(lambda m: " " + render_latex(m.group(1), vocabulary) + " ", text)
text = _LEADING_HASHES.sub("", text)
text = _MD_BOLD.sub(r"\1", text)
text = _MD_ITALIC.sub(r"\1", text)
return text.replace("`", "")
def _clean_heading(raw: str) -> tuple[str, int | None]:
"""A `title` block's markdown → (plain heading text, level from '#' depth)."""
m = _LEADING_HASHES.match(raw)
level = len(m.group(1)) if m else None
text = _LEADING_HASHES.sub("", raw)
text = _MD_BOLD.sub(r"\1", text)
text = _MD_ITALIC.sub(r"\1", text)
return text.strip(), level
def _bbox(block: dict[str, Any]) -> list[int] | None:
keys = ("top_left_x", "top_left_y", "bottom_right_x", "bottom_right_y")
if all(k in block and block[k] is not None for k in keys):
return [int(block[k]) for k in keys]
return None
def _blocks_with_pages(response: dict[str, Any]) -> Iterator[tuple[int, dict[str, Any]]]:
"""Flatten pages[].blocks into (page_idx, block), page order preserved."""
for page in response.get("pages", []):
pidx = int(page.get("index", 0))
for block in page.get("blocks") or []:
yield pidx, block
def _image_descriptions(response: dict[str, Any]) -> dict[str, str]:
"""image_id -> description, from the page-level annotations (bbox_annotation).
Empty when the OCR call did not request annotations; description then stays
None, exactly as it does on MinerU's `pipeline` backend.
"""
out: dict[str, str] = {}
for page in response.get("pages", []):
for im in page.get("images") or []:
ann = im.get("image_annotation")
iid = im.get("id")
if iid and ann:
out[iid] = _unwrap_description(ann)
return out
def _unwrap_description(annotation: Any) -> str:
"""Pull the prose out of an annotation payload.
Mistral returns bbox annotations JSON-WRAPPED — literally
`{"description": "The diagram illustrates…"}` — and 0.3.x stored that string
raw, braces and all, so every consumer got JSON where it expected a sentence.
Unwrapped here, at the one place that reads the field.
Never raises: anything unparseable falls back to the string form, because a
slightly ugly description is worth more than a failed parse.
"""
if isinstance(annotation, dict):
value = annotation.get("description") or annotation.get("text")
return str(value) if value else json.dumps(annotation, ensure_ascii=False)
if isinstance(annotation, str):
text = annotation.strip()
if text.startswith("{"):
try:
parsed = json.loads(text)
except (ValueError, TypeError):
return text
if isinstance(parsed, dict):
value = parsed.get("description") or parsed.get("text")
if value:
return str(value)
return text
return text
return str(annotation)
def _block_image_id(block: dict[str, Any]) -> str | None:
"""The image this block refers to, preferring the id in its own markdown.
⚠️ **`block["image_id"]` is not trustworthy.** Measured on the Open Pit
textbook: page 6 carries two image blocks, and the FIRST one's `content` reads
`![img-1.jpeg](img-1.jpeg)` while its `image_id` field says **`img-2.jpeg`** —
so both blocks claim the same image. Keying on `image_id` therefore resolved
both to img-2's bytes, gave two different figures one content hash, and lost
img-1 entirely (7 files written for 8 figures).
The `content` markdown is the document's own reference and was correct in
every case, so it wins; `image_id` is the fallback for a block whose content
carries no reference. This also corrects an earlier diagnosis: the ids are
globally unique across the response (`img-0` … `img-7`), so the duplication
was never page-local naming — it was this mislabelled field.
"""
match = _MD_IMAGE_TARGET.search(block.get("content") or "")
if match and match.group(1).strip():
return match.group(1).strip()
return block.get("image_id")
def _image_payloads(response: dict[str, Any]) -> dict[str, str]:
"""image_id -> base64 payload. Empty when `include_image_base64` was off.
0.3.x never read these: the bytes were returned by the API, recorded nowhere,
and `Chunk.images` pointed at filenames that did not exist on disk. 0.4.0
content-addresses and persists them.
"""
out: dict[str, str] = {}
for page in response.get("pages", []):
for im in page.get("images") or []:
iid, payload = im.get("id"), im.get("image_base64")
if iid and payload:
out[iid] = payload
return out
def _chunk_by_page(
blocks: list[tuple[int, dict[str, Any]]],
descriptions: dict[str, str],
chapter_per_page: dict[int, str],
doc_id: str,
) -> list[Chunk]:
"""Page granularity: all prose on a page → one chunk; tables/images their own.
Recall-only. A page chunk has no single heading, so `heading_path` /
`section_no` stay empty and the extraction outline degrades — section
granularity (the default) is the real path.
"""
def chapters(page: int) -> list[str]:
return [chapter_per_page[page]] if page in chapter_per_page else []
chunks: list[Chunk] = []
by_page: dict[int, list[tuple[int, dict[str, Any]]]] = {}
for i, (pidx, block) in enumerate(blocks):
by_page.setdefault(pidx, []).append((i, block))
for pidx in sorted(by_page):
pieces: list[str] = []
latex: list[str] = []
src: list[int] = []
has_formula = False
prose_bbox: list[int] | None = None
for i, block in by_page[pidx]:
kind = block.get("type")
if kind in _DISCARDED or kind == "header":
continue
if kind == "table":
c = Chunk(
chunk_id=f"{doc_id}::{len(chunks):04d}", doc_id=doc_id,
kind="table", text=render_table(block.get("content") or ""),
page_idx=pidx, page_idxs=[pidx], chapters=chapters(pidx),
is_tabular=True, table_html=block.get("content") or None,
source_items=[i], bbox=_bbox(block),
)
if c.text or c.images:
chunks.append(c)
continue
if kind == "image":
iid = block.get("image_id")
chunks.append(Chunk(
chunk_id=f"{doc_id}::{len(chunks):04d}", doc_id=doc_id,
kind="image", text="", page_idx=pidx, page_idxs=[pidx],
chapters=chapters(pidx),
generated_description=descriptions.get(iid) if iid else None,
source_items=[i], bbox=_bbox(block),
images=[iid] if iid else [],
))
continue
if kind == "equation":
raw = (block.get("content") or "").strip().strip("$").strip()
has_formula = True
src.append(i)
if raw:
latex.append(raw)
rendered = render_latex(raw)
if rendered:
pieces.append(rendered)
continue
# text / title / list / caption / signature -> prose. A numbered
# heading is NOT a boundary in page mode; it stays as prose text.
text = _strip_markdown(block.get("content") or "").rstrip()
if not text.strip():
continue
if prose_bbox is None:
prose_bbox = _bbox(block)
src.append(i)
pieces.append(text)
if pieces:
chunks.append(Chunk(
chunk_id=f"{doc_id}::{len(chunks):04d}", doc_id=doc_id,
kind="equation" if has_formula else "text",
text="\n\n".join(pieces).strip(),
page_idx=pidx, page_idxs=[pidx], chapters=chapters(pidx),
has_formula=has_formula, latex=latex, source_items=src,
bbox=prose_bbox,
))
return chunks
def normalise_mistral(
response: dict[str, Any],
doc_id: str,
granularity: str = "section",
asset_sink: AssetSink | None = None,
) -> list[Chunk]:
"""Mistral OCR response (include_blocks=True) → chunks.
`granularity` decides how prose is grouped, and only that:
- "section" (default): a numbered heading opens a new chunk, mirroring
normalize.py. This is the default because the extraction half derives
the document `outline` from `heading_path` and drives `source_wording`
from `heading` — page grouping leaves both EMPTY (a page chunk has no
single heading), which silently degrades a locked product behaviour.
- "page": all prose on a page becomes ONE chunk. Higher raw term recall
(bigger context per NER window) but no per-chunk heading, so only fit
for a recall-only measurement, never for the real pipeline.
The section/page recall gap (0.878 vs 0.951 on the BUMA fixture) is the NER
filter's, not content: every gold term is present in the text either way, and
closing it is the extraction side's lever (span-filter overlap resolution).
In BOTH modes tables and images are their OWN chunks, so `table_html` stays
one-table-per-chunk regardless. Switching modes is a one-word change and
never touches the extraction half — the artifact shape is identical.
No `page_vocabulary` parameter: Mistral does not spell formulas out one
character at a time, so the word-boundary recovery that MinerU needs has
nothing to do here.
"""
if granularity not in {"page", "section"}:
raise ValueError(f"granularity must be 'page' or 'section', got {granularity!r}")
blocks = list(_blocks_with_pages(response))
if not blocks:
raise ValueError(
"no blocks in Mistral response — call ocr.process with include_blocks=True"
)
descriptions = _image_descriptions(response)
# Chapter context per page, from the running header (type 'header').
chapter_per_page: dict[int, str] = {}
for pidx, b in blocks:
if b.get("type") == "header":
txt = _strip_markdown(b.get("content") or "").strip()
if txt:
chapter_per_page.setdefault(pidx, txt)
# --- page mode: recall-only; a page chunk has no single heading ---
if granularity == "page":
# ⚠️ FENCED OFF AT 0.4.0, deliberately, rather than given the asset
# treatment. Page mode emits 0.3.x-shaped chunks — images standalone with
# empty text, no `assets` — so letting it run would quietly produce an
# artifact that CLAIMS schema_version 0.4.0 while having the old shape.
# Two normalisers silently diverging is exactly how the `type: "list"`
# content drop survived unnoticed and cost five gold terms.
#
# Page mode exists only for a recall-only measurement (it nulls
# `heading_path`, which the outline and `source_wording` both need), so it
# is not worth duplicating the asset logic into. Restore by porting the
# image/table branches from the section path below.
raise NotImplementedError(
"granularity='page' is not supported at schema 0.4.0: it would emit "
"0.3.x-shaped chunks (standalone image chunks, no assets) under a "
"0.4.0 version stamp. Use granularity='section', or port the asset "
"handling into _chunk_by_page first."
)
# --- section mode (default): a numbered heading opens a new chunk (mirrors normalize.py) ---
chunks: list[Chunk] = []
running: Chunk | None = None
pieces: list[str] = []
stack: list[tuple[int, str]] = []
payloads = _image_payloads(response)
# A table's asset has to be added to the PROSE chunk that precedes it, but
# that chunk is already closed and appended by then — so the back-reference is
# queued and resolved once, after the loop.
_pending_back_refs: list[tuple[str, Asset]] = []
# A printed caption arrives as its own `caption` block AFTER its image block.
# The figure waiting for one is held here; the caption is attached to the asset
# AND left in the prose text, because it is verbatim document content.
_pending_caption_for: list[Asset] = []
def push_heading(level: int, text: str) -> None:
while stack and stack[-1][0] >= level:
stack.pop()
stack.append((level, text))
def close() -> None:
nonlocal running, pieces
if running is not None:
running.text = "\n\n".join(pieces).strip()
if running.text or running.assets or running.images:
running.page_idxs = sorted(set(running.page_idxs))
chunks.append(running)
running, pieces = None, []
def open_chunk(
kind: str, page: int, section_no: str | None = None, heading: str | None = None
) -> Chunk:
return Chunk(
chunk_id=f"{doc_id}::{len(chunks):04d}",
doc_id=doc_id, kind=kind, text="",
page_idx=page, page_idxs=[page],
section_no=section_no, heading=heading,
chapters=[chapter_per_page[page]] if page in chapter_per_page else [],
heading_path=[t for _, t in stack],
)
for i, (page, block) in enumerate(blocks):
kind = block.get("type")
if kind in _DISCARDED or kind == "header":
continue
if kind == "table":
# Tables stay their OWN chunk (0.4.0 kept this deliberately): unlike a
# figure, a table's linearised text carries real terms the filter
# finds — 74.6% of the Komatsu manual's top-3 evidence is tabular. What
# 0.4.0 adds is reachability: an `Asset` plus a two-way pointer, so the
# prose that introduces a table can be followed to it and back.
prose_before = running.chunk_id if running is not None else None
close()
c = open_chunk("table", page)
c.text = render_table(block.get("content") or "")
c.is_tabular = True
c.table_html = block.get("content") or None
c.source_items = [i]
c.bbox = _bbox(block)
table_asset = Asset(
# No bytes to hash, so the id is derived from stable document
# coordinates instead — deterministic across re-parses, which is
# what a durable pointer needs.
asset_id=synthetic_asset_id(doc_id, c.chunk_id, str(i)),
kind="table",
page_idx=page,
bbox=_bbox(block),
)
c.assets = [table_asset]
if prose_before is not None:
c.referenced_by = [prose_before]
_pending_back_refs.append((prose_before, table_asset))
if c.text or c.assets:
chunks.append(c)
continue
if kind == "image":
# ── 0.4.0: a figure is NO LONGER ITS OWN CHUNK. ────────────────────
# It used to be, with an EMPTY `text`, which made it unreachable by
# construction: no text -> no term mentions -> joins no cluster ->
# never selected as evidence. Measured across three documents: 13
# figures, every one described by a vision model we pay for, and ZERO
# appearing in any cluster's evidence. Meanwhile the prose saying
# "Figure 1.1" had its `![](…)` reference stripped, so nothing
# connected the sentence to the picture.
#
# Now the reference is folded into the RUNNING prose chunk at its
# reading-order position — exactly where the document put it — and the
# figure travels as an `Asset` on that chunk.
iid = _block_image_id(block)
identifier, storage_key = store_asset(
asset_sink, iid, payloads.get(iid) if iid else None
)
# Fall back to document coordinates when the parser returned no bytes,
# so a figure is still referenced rather than silently vanishing.
if identifier is None:
identifier = synthetic_asset_id(doc_id, str(page), str(i))
asset = Asset(
asset_id=identifier,
kind="figure",
storage_key=storage_key,
page_idx=page,
bbox=_bbox(block),
description=descriptions.get(iid) if iid else None,
description_source=_DESCRIPTION_SOURCE if iid else None,
)
# P5 edge case: a figure can precede any prose on its page (Open Pit
# p.3 is close to this shape). Open a chunk rather than drop the asset.
if running is None:
running = open_chunk("text", page)
running.source_items.append(i)
running.page_idxs.append(page)
running.assets.append(asset)
# Deprecated alias, one version only.
if iid:
running.images.append(iid)
# The reference itself, inline in the prose. `asset://` rather than the
# parser's filename because the filename is page-local and collides.
pieces.append(f"![](asset://{identifier})")
_pending_caption_for.append(asset)
continue
if kind == "equation":
latex = (block.get("content") or "").strip().strip("$").strip()
if running is None:
running = open_chunk("text", page)
running.has_formula = True
running.source_items.append(i)
running.page_idxs.append(page)
if latex:
running.latex.append(latex) # raw, for the formula branch
rendered = render_latex(latex)
if rendered:
pieces.append(rendered) # prose, so NER can find it
continue
raw = block.get("content") or ""
# `list` and `caption` are ALWAYS content, never a section boundary: a
# bullet or a caption can look like a numbered heading, and letting one
# open a section would split a list from the paragraph introducing it
# (see normalize.py's list note).
if kind in {"list", "caption"}:
text = _strip_markdown(raw).rstrip()
if text.strip():
if running is None:
running = open_chunk("text", page)
running.bbox = _bbox(block)
running.source_items.append(i)
running.page_idxs.append(page)
pieces.append(text)
# A caption directly after a figure names that figure. Attach it to
# the asset so a consumer can render "Figure 1.1 …" beside the
# image — and LEAVE it in `pieces` too, because it is printed in
# the document: verbatim, quotable as evidence, span-checkable.
# Copying it is the point; moving it would silently delete
# evidence that survived 17/17 times on the Open Pit textbook.
if kind == "caption" and _pending_caption_for:
_pending_caption_for.pop(0).caption = text.strip()
continue
# `text` / `title` / `signature`: a NUMBERED line may open a section.
# normalize._heading is applied to BOTH text and title items on MinerU —
# here too, because Mistral labels many numbered headings as `text`, not
# `title`. Checking only `title` left whole sections unopened and every
# equation piled into one chunk. A title's level comes from its `#`
# depth; a text line's from numbering depth (fallback inside _heading).
if kind == "title":
htext, level = _clean_heading(raw)
else:
htext, level = _strip_markdown(raw).strip(), None
number, heading, hlevel = _heading(
{"type": "title", "text": htext, "text_level": level}
)
if heading is not None:
if any(htext == t for _, t in stack): # reprinted breadcrumb
if running is not None:
running.page_idxs.append(page)
continue
close()
push_heading(hlevel or 1, heading)
running = open_chunk("text", page, section_no=number, heading=heading)
running.source_items = [i]
running.bbox = _bbox(block)
# heading line NOT copied into text — lives in `heading`. See
# normalize.py for the measured reason (mention_count inflation).
continue
text = _strip_markdown(raw).rstrip()
if not text.strip():
continue
if running is None:
running = open_chunk("text", page)
running.bbox = _bbox(block)
running.source_items.append(i)
running.page_idxs.append(page)
chapter = chapter_per_page.get(page)
if chapter and chapter not in running.chapters:
running.chapters.append(chapter)
pieces.append(text) # verbatim (markup only removed)
if sum(len(x) for x in pieces) >= MAX_CHUNK_CHARS:
continued_from = running
close()
running = open_chunk("text", page,
section_no=continued_from.section_no,
heading=continued_from.heading)
close()
# Two-way table pointers, resolved once the loop is done. The table chunk
# already names the prose chunk in `referenced_by`; this adds the table's asset
# to that prose chunk so the link is navigable from the prose side too — the
# direction a reader, and a model reading evidence, actually travels. Deferred
# rather than done inline because the prose chunk is already closed and
# appended by the time its table is reached.
by_id = {c.chunk_id: c for c in chunks}
for prose_id, table_asset in _pending_back_refs:
host = by_id.get(prose_id)
if host is not None and not any(
a.asset_id == table_asset.asset_id for a in host.assets
):
host.assets.append(table_asset)
return chunks