ishaq101's picture sofhiaazzhr's picture
/feat knowledge management (#20)
b68816f
Raw History Blame Contribute Delete
12.1 kB
"""Render MinerU's markup back into readable prose for `Chunk.text`.
Why this exists — measured, not assumed. Running the same extraction pipeline,
same document, same gold set, changing only the parse:
PyMuPDF (plain text) recall 0.8537 35/41
MinerU, raw markup in text recall 0.7561 31/41
MinerU, formula + table rendered recall 0.8293 34/41
The term filter is an NER model reading prose. MinerU writes formulas with every
character spaced out — `{ \\mathrm { P u r c h a s i n g ~ c o s t s } }` — and
tables as HTML. Neither yields a single mention, even for terms the same model
finds easily in plain text.
The markup is not discarded: `Chunk.latex` and `Chunk.table_html` keep it
verbatim, because the formula branch needs exactly that form.
"""
from __future__ import annotations
import html
import re
# --- LaTeX ---------------------------------------------------------------
_DROP_COMMANDS = re.compile(
r"\\(?:qquad|quad|left|right|displaystyle|limits|nolimits)\b|\\[!,;:]"
)
_WRAPPERS = re.compile(r"\\(?:mathrm|mathbf|mathit|mathsf|text|textrm|operatorname)\s*")
# Operators map to non-letter glyphs on purpose. A letter here (e.g. "x" for
# \times) would itself look like a spelled-out single character and get merged
# into the token beside it — "C_p x T" becoming "C_p xT".
_SYMBOLS = {
r"\times": "×", r"\cdot": "·", r"\div": "/", r"\pm": "±",
r"\leq": "≤", r"\geq": "≥", r"\neq": "≠", r"\approx": "≈",
r"\%": "%", r"\$": "$", r"\&": "&",
}
# "P u r c h a s i n g" -> "Purchasing", "1 0 0" -> "100". Runs of >= 2 single
# characters; letters and digits are joined separately so "C 5" is left alone.
# Subscripts are folded first (below), so a real variable like `C_p` is already
# one token and never gets swallowed into a neighbouring word.
#
# The boundaries are word-character classes, not `\S`. Two reasons, both found
# on real equations:
# - `\frac` rewriting INTRODUCES parentheses, and `(?<!\S)` refuses to match a
# run touching one, stranding a letter at each end:
# "(I N P R H o u r s)" -> "(I NPRHour s)".
# - the trailing side lets `_`/`^` follow, so a run whose last letter carries a
# subscript is not cut short: "Q t y_activity" -> "Qty_activity".
_SPACED_RUN = re.compile(
r"(?<![A-Za-z0-9_])(?:[A-Za-z]\s+){1,}[A-Za-z](?![A-Za-z0-9])"
r"|(?<![A-Za-z0-9_])(?:\d\s+){1,}\d(?![A-Za-z0-9])"
)
# Wrappers that take a braced argument. Handled by brace matching rather than
# `_WRAPPERS` below, because merely deleting the command name leaves its braces
# behind — and those braces are what stopped `\frac` from matching.
_BRACED_WRAPPER = re.compile(
r"\\(?:mathrm|mathbf|mathit|mathsf|mathtt|text|textrm|textbf|operatorname)\s*(?=\{)"
)
# `~` is a non-breaking SPACE in LaTeX, i.e. a real word boundary. It has to
# survive the whitespace pass, otherwise "c o s t s" after it merges into the
# preceding word and "Purchasing costs" becomes "Purchasingcosts".
_WORD_GAP = "\x00"
def _join_spaced(text: str) -> str:
"""Join runs of characters spelled out one by one: 'i j' -> 'ij'."""
return _SPACED_RUN.sub(lambda m: m.group(0).replace(" ", ""), text)
# Minimum token length eligible for splitting. Measured on real documents: 6, 4
# and 3 give identical results on equations, and 3 additionally recovers `MOHHx`
# / `xPA_A` inside table cells. Dropping to 2 starts doing DAMAGE — `PA_ij`
# splits into `PA_i j`, because `i` and `j` are both valid words on that page.
MIN_TOKEN_TO_SPLIT = 3
_MAX_WORD_LEN = 24 # a sane single-word length; bounds the DP work
_TOKEN = re.compile(r"[A-Za-z][A-Za-z0-9]*")
def _segment(token: str, vocabulary: frozenset[str]) -> list[str] | None:
"""Split `token` into words that are ALL present in `vocabulary`.
Returns None when no valid split exists, OR when more than one shortest
split is equally valid. Ambiguity is left alone — this module does not
guess, and a token that cannot be recovered with certainty is better left
damaged than repaired into the wrong form with no trace.
"""
n = len(token)
# dp[i] = (fewest words for token[:i], how many ways reach that, the parts)
dp: list[tuple[int, int, list[str]] | None] = [None] * (n + 1)
dp[0] = (0, 1, [])
for i in range(1, n + 1):
best: int | None = None
ways = 0
parts: list[str] = []
for j in range(max(0, i - _MAX_WORD_LEN), i):
prev = dp[j]
if prev is None or token[j:i] not in vocabulary:
continue
prev_len, prev_ways, prev_words = prev
candidate = prev_len + 1
if best is None or candidate < best:
best, ways, parts = candidate, prev_ways, prev_words + [token[j:i]]
elif candidate == best:
ways += prev_ways
if best is not None:
dp[i] = (best, ways, parts)
result = dp[n]
if result is None:
return None
_, ways, parts = result
if ways != 1 or len(parts) < 2:
return None
return parts
def restore_word_boundaries(text: str, vocabulary: frozenset[str] | None) -> str:
"""Put back the spaces lost when a character-spaced run was joined up.
MinerU writes formulas one character at a time (`T o t a l H o u r s`), and
once those are joined the word boundaries cannot be recovered from the LaTeX
itself: `Total Hours` and `TotalHours` become indistinguishable. A
multiplication sign written as the letter `x` fuses in too, because it looks
exactly like any other spaced character.
The source PDF's text layer STILL holds those boundaries, so the answer is
read from the document rather than guessed. A token already present as a word
in the source is never touched — which means a document that writes its
formulas correctly skips this path entirely.
"""
if not vocabulary:
return text
replacements: dict[str, str] = {}
for token in set(_TOKEN.findall(text)):
if len(token) < MIN_TOKEN_TO_SPLIT or token in vocabulary:
continue
parts = _segment(token, vocabulary)
if parts:
replacements[token] = " ".join(parts)
for token, joined in sorted(replacements.items(), key=lambda kv: -len(kv[0])):
text = re.sub(rf"(?<![A-Za-z0-9]){re.escape(token)}(?![A-Za-z0-9])", joined, text)
return text
def _braced_argument(s: str, i: int) -> tuple[str | None, int]:
"""The BALANCED brace contents starting at `s[i] == '{'`.
A regex `\\{([^{}]*)\\}` cannot be used here: real arguments contain nested
braces (`\\mathrm{...}`, `\\sum _{ij}`), and that pattern fails to match
silently rather than raising.
"""
depth = 0
for j in range(i, len(s)):
if s[j] == "{":
depth += 1
elif s[j] == "}":
depth -= 1
if depth == 0:
return s[i + 1:j], j + 1
return None, i
def _unwrap_braced(s: str) -> str:
r"""`\mathrm{X}` -> `X`, consuming its braces along with it."""
for _ in range(8):
m = _BRACED_WRAPPER.search(s)
if not m:
break
inner, end = _braced_argument(s, m.end())
if inner is None:
break
s = s[:m.start()] + " " + inner + " " + s[end:]
return s
def _render_frac(s: str) -> str:
r"""`\frac{a}{b}` -> `(a) / (b)`, brace-aware and recursive."""
for _ in range(8):
m = re.search(r"\\d?frac\s*(?=\{)", s)
if not m:
break
numerator, after = _braced_argument(s, s.index("{", m.end() - 1))
if numerator is None:
break
denom_start = s.find("{", after)
if denom_start == -1:
break
denominator, end = _braced_argument(s, denom_start)
if denominator is None:
break
s = (s[:m.start()] + " (" + _render_frac(numerator) + ") / ("
+ _render_frac(denominator) + ") " + s[end:])
return s
def render_latex(latex: str, vocabulary: frozenset[str] | None = None) -> str:
"""LaTeX -> a readable line. Best effort: it feeds a term filter, not a parser.
`vocabulary` is the source page's own word list, used only to put back word
boundaries this function cannot recover on its own — see
`restore_word_boundaries`. Omitting it reproduces the previous behaviour
exactly, so existing callers are unaffected.
"""
s = latex.replace("$$", " ").replace("$", " ")
s = s.replace("~", _WORD_GAP)
# array/matrix scaffolding carries no meaning for a term filter
s = re.sub(r"\\(?:begin|end)\s*\{[^{}]*\}", " ", s)
s = re.sub(r"\{\s*(?:[rlc|]\s*){1,8}\}", " ", s) # column spec, e.g. { r l }
s = s.replace("\\\\", " ; ").replace("&", " ")
# ORDER MATTERS, and this is the ordering decision: unwrap the braced
# wrappers BEFORE touching `\frac`.
#
# `\frac`'s arguments in real documents are not brace-free — the BUMA
# standard writes `\frac { \mathrm { T o t a l } ~ H o u r s - ... } { ... }`.
# With the wrappers still in place the old `\{([^{}]*)\}` pattern could not
# match, so the `\frac` was never rewritten; it then fell through to the
# generic "drop any leftover command" pass and the DIVISION WAS SILENTLY
# DELETED. `PA = (Total - Breakdown) / Total` came out as
# `PA = Total Hours - BreakdownTotal Hours`, which reads as a product and
# states something the document does not. It hit 4 of 11 equations —
# PA, PA_ij, UA_ij, PTY_ij.
s = _unwrap_braced(s)
# fold sub/superscripts, so `C _ { p }` becomes one token `C_p`. The content
# is joined too: `_ { i j }` -> `_ij`, otherwise the space inside survives
# and splits the token later ("_i j").
s = re.sub(r"\s*_\s*\{\s*([^{}]*?)\s*\}", lambda m: "_" + _join_spaced(m.group(1)), s)
s = re.sub(r"\s*\^\s*\{\s*([^{}]*?)\s*\}", lambda m: "^" + _join_spaced(m.group(1)), s)
s = re.sub(r"\s*_\s*([A-Za-z0-9])", r"_\1", s)
s = re.sub(r"\s*\^\s*([A-Za-z0-9])", r"^\1", s)
s = _render_frac(s)
s = re.sub(r"\\sqrt\s*\{([^{}]*)\}", r"sqrt(\1)", s)
for k, v in _SYMBOLS.items():
s = s.replace(k, f" {v} ")
s = _WRAPPERS.sub(" ", s) # leftover wrappers with no braces, e.g. `\mathrm x`
s = _DROP_COMMANDS.sub(" ", s)
s = re.sub(r"\\[A-Za-z]+", " ", s) # any command left over
s = s.replace("{", " ").replace("}", " ").replace("\\", " ")
s = re.sub(r"[ \t\r\n]+", " ", s).strip()
# now that spacing is uniform, rejoin the spelled-out words and numbers
s = _join_spaced(s)
s = s.replace(_WORD_GAP, " ")
s = re.sub(r"\s+", " ", s).strip()
# Last, once normal spacing is back: restore the word boundaries that the
# LaTeX itself cannot tell us about.
return restore_word_boundaries(s, vocabulary)
# --- HTML tables ---------------------------------------------------------
_CELL_END = re.compile(r"</\s*(?:td|th)\s*>", re.I)
_ROW_END = re.compile(r"</\s*tr\s*>", re.I)
_TAG = re.compile(r"<[^>]+>")
# Table cells routinely carry inline LaTeX ("$\frac{kVA}{...}$"). Stripping the
# HTML alone leaves that markup sitting in the prose, which is the same problem
# the formula rendering exists to solve — found in a real table on the first run.
_INLINE_MATH = re.compile(r"\$([^$]{1,400})\$")
def render_table(table_html: str, vocabulary: frozenset[str] | None = None) -> str:
"""HTML table -> plain rows, cells separated by ' | '.
Keeps the reading order a person would use, so a term appearing only in a
table header is still a mention.
"""
s = _ROW_END.sub("\n", table_html)
s = _CELL_END.sub(" | ", s)
s = _TAG.sub(" ", s)
s = html.unescape(s)
s = _INLINE_MATH.sub(lambda m: " " + render_latex(m.group(1), vocabulary) + " ", s)
rows = []
for raw in s.split("\n"):
cleaned = re.sub(r"[ \t]+", " ", raw).strip().strip("|").strip()
if cleaned:
rows.append(re.sub(r"\s*\|\s*", " | ", cleaned))
return "\n".join(rows)