File size: 12,102 Bytes
b68816f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
"""Render MinerU's markup back into readable prose for `Chunk.text`.

Why this exists β€” measured, not assumed. Running the same extraction pipeline,
same document, same gold set, changing only the parse:

    PyMuPDF (plain text)                recall 0.8537   35/41
    MinerU, raw markup in text          recall 0.7561   31/41
    MinerU, formula + table rendered     recall 0.8293   34/41

The term filter is an NER model reading prose. MinerU writes formulas with every
character spaced out β€” `{ \\mathrm { P u r c h a s i n g ~ c o s t s } }` β€” and
tables as HTML. Neither yields a single mention, even for terms the same model
finds easily in plain text.

The markup is not discarded: `Chunk.latex` and `Chunk.table_html` keep it
verbatim, because the formula branch needs exactly that form.
"""

from __future__ import annotations

import html
import re

# --- LaTeX ---------------------------------------------------------------

_DROP_COMMANDS = re.compile(
    r"\\(?:qquad|quad|left|right|displaystyle|limits|nolimits)\b|\\[!,;:]"
)
_WRAPPERS = re.compile(r"\\(?:mathrm|mathbf|mathit|mathsf|text|textrm|operatorname)\s*")
# Operators map to non-letter glyphs on purpose. A letter here (e.g. "x" for
# \times) would itself look like a spelled-out single character and get merged
# into the token beside it β€” "C_p x T" becoming "C_p xT".
_SYMBOLS = {
    r"\times": "Γ—", r"\cdot": "Β·", r"\div": "/", r"\pm": "Β±",
    r"\leq": "≀", r"\geq": "β‰₯", r"\neq": "β‰ ", r"\approx": "β‰ˆ",
    r"\%": "%", r"\$": "$", r"\&": "&",
}

# "P u r c h a s i n g" -> "Purchasing", "1 0 0" -> "100". Runs of >= 2 single
# characters; letters and digits are joined separately so "C 5" is left alone.
# Subscripts are folded first (below), so a real variable like `C_p` is already
# one token and never gets swallowed into a neighbouring word.
#
# The boundaries are word-character classes, not `\S`. Two reasons, both found
# on real equations:
#   - `\frac` rewriting INTRODUCES parentheses, and `(?<!\S)` refuses to match a
#     run touching one, stranding a letter at each end:
#     "(I N P R H o u r s)" -> "(I NPRHour s)".
#   - the trailing side lets `_`/`^` follow, so a run whose last letter carries a
#     subscript is not cut short: "Q t y_activity" -> "Qty_activity".
_SPACED_RUN = re.compile(
    r"(?<![A-Za-z0-9_])(?:[A-Za-z]\s+){1,}[A-Za-z](?![A-Za-z0-9])"
    r"|(?<![A-Za-z0-9_])(?:\d\s+){1,}\d(?![A-Za-z0-9])"
)

# Wrappers that take a braced argument. Handled by brace matching rather than
# `_WRAPPERS` below, because merely deleting the command name leaves its braces
# behind β€” and those braces are what stopped `\frac` from matching.
_BRACED_WRAPPER = re.compile(
    r"\\(?:mathrm|mathbf|mathit|mathsf|mathtt|text|textrm|textbf|operatorname)\s*(?=\{)"
)

# `~` is a non-breaking SPACE in LaTeX, i.e. a real word boundary. It has to
# survive the whitespace pass, otherwise "c o s t s" after it merges into the
# preceding word and "Purchasing costs" becomes "Purchasingcosts".
_WORD_GAP = "\x00"


def _join_spaced(text: str) -> str:
    """Join runs of characters spelled out one by one: 'i j' -> 'ij'."""
    return _SPACED_RUN.sub(lambda m: m.group(0).replace(" ", ""), text)


# Minimum token length eligible for splitting. Measured on real documents: 6, 4
# and 3 give identical results on equations, and 3 additionally recovers `MOHHx`
# / `xPA_A` inside table cells. Dropping to 2 starts doing DAMAGE β€” `PA_ij`
# splits into `PA_i j`, because `i` and `j` are both valid words on that page.
MIN_TOKEN_TO_SPLIT = 3
_MAX_WORD_LEN = 24       # a sane single-word length; bounds the DP work

_TOKEN = re.compile(r"[A-Za-z][A-Za-z0-9]*")


def _segment(token: str, vocabulary: frozenset[str]) -> list[str] | None:
    """Split `token` into words that are ALL present in `vocabulary`.

    Returns None when no valid split exists, OR when more than one shortest
    split is equally valid. Ambiguity is left alone β€” this module does not
    guess, and a token that cannot be recovered with certainty is better left
    damaged than repaired into the wrong form with no trace.
    """
    n = len(token)
    # dp[i] = (fewest words for token[:i], how many ways reach that, the parts)
    dp: list[tuple[int, int, list[str]] | None] = [None] * (n + 1)
    dp[0] = (0, 1, [])

    for i in range(1, n + 1):
        best: int | None = None
        ways = 0
        parts: list[str] = []
        for j in range(max(0, i - _MAX_WORD_LEN), i):
            prev = dp[j]
            if prev is None or token[j:i] not in vocabulary:
                continue
            prev_len, prev_ways, prev_words = prev
            candidate = prev_len + 1
            if best is None or candidate < best:
                best, ways, parts = candidate, prev_ways, prev_words + [token[j:i]]
            elif candidate == best:
                ways += prev_ways
        if best is not None:
            dp[i] = (best, ways, parts)

    result = dp[n]
    if result is None:
        return None
    _, ways, parts = result
    if ways != 1 or len(parts) < 2:
        return None
    return parts


def restore_word_boundaries(text: str, vocabulary: frozenset[str] | None) -> str:
    """Put back the spaces lost when a character-spaced run was joined up.

    MinerU writes formulas one character at a time (`T o t a l   H o u r s`), and
    once those are joined the word boundaries cannot be recovered from the LaTeX
    itself: `Total Hours` and `TotalHours` become indistinguishable. A
    multiplication sign written as the letter `x` fuses in too, because it looks
    exactly like any other spaced character.

    The source PDF's text layer STILL holds those boundaries, so the answer is
    read from the document rather than guessed. A token already present as a word
    in the source is never touched β€” which means a document that writes its
    formulas correctly skips this path entirely.
    """
    if not vocabulary:
        return text

    replacements: dict[str, str] = {}
    for token in set(_TOKEN.findall(text)):
        if len(token) < MIN_TOKEN_TO_SPLIT or token in vocabulary:
            continue
        parts = _segment(token, vocabulary)
        if parts:
            replacements[token] = " ".join(parts)

    for token, joined in sorted(replacements.items(), key=lambda kv: -len(kv[0])):
        text = re.sub(rf"(?<![A-Za-z0-9]){re.escape(token)}(?![A-Za-z0-9])", joined, text)
    return text


def _braced_argument(s: str, i: int) -> tuple[str | None, int]:
    """The BALANCED brace contents starting at `s[i] == '{'`.

    A regex `\\{([^{}]*)\\}` cannot be used here: real arguments contain nested
    braces (`\\mathrm{...}`, `\\sum _{ij}`), and that pattern fails to match
    silently rather than raising.
    """
    depth = 0
    for j in range(i, len(s)):
        if s[j] == "{":
            depth += 1
        elif s[j] == "}":
            depth -= 1
            if depth == 0:
                return s[i + 1:j], j + 1
    return None, i


def _unwrap_braced(s: str) -> str:
    r"""`\mathrm{X}` -> `X`, consuming its braces along with it."""
    for _ in range(8):
        m = _BRACED_WRAPPER.search(s)
        if not m:
            break
        inner, end = _braced_argument(s, m.end())
        if inner is None:
            break
        s = s[:m.start()] + " " + inner + " " + s[end:]
    return s


def _render_frac(s: str) -> str:
    r"""`\frac{a}{b}` -> `(a) / (b)`, brace-aware and recursive."""
    for _ in range(8):
        m = re.search(r"\\d?frac\s*(?=\{)", s)
        if not m:
            break
        numerator, after = _braced_argument(s, s.index("{", m.end() - 1))
        if numerator is None:
            break
        denom_start = s.find("{", after)
        if denom_start == -1:
            break
        denominator, end = _braced_argument(s, denom_start)
        if denominator is None:
            break
        s = (s[:m.start()] + " (" + _render_frac(numerator) + ") / ("
             + _render_frac(denominator) + ") " + s[end:])
    return s


def render_latex(latex: str, vocabulary: frozenset[str] | None = None) -> str:
    """LaTeX -> a readable line. Best effort: it feeds a term filter, not a parser.

    `vocabulary` is the source page's own word list, used only to put back word
    boundaries this function cannot recover on its own β€” see
    `restore_word_boundaries`. Omitting it reproduces the previous behaviour
    exactly, so existing callers are unaffected.
    """
    s = latex.replace("$$", " ").replace("$", " ")
    s = s.replace("~", _WORD_GAP)

    # array/matrix scaffolding carries no meaning for a term filter
    s = re.sub(r"\\(?:begin|end)\s*\{[^{}]*\}", " ", s)
    s = re.sub(r"\{\s*(?:[rlc|]\s*){1,8}\}", " ", s)   # column spec, e.g. { r l }
    s = s.replace("\\\\", " ; ").replace("&", " ")

    # ORDER MATTERS, and this is the ordering decision: unwrap the braced
    # wrappers BEFORE touching `\frac`.
    #
    # `\frac`'s arguments in real documents are not brace-free β€” the BUMA
    # standard writes `\frac { \mathrm { T o t a l } ~ H o u r s - ... } { ... }`.
    # With the wrappers still in place the old `\{([^{}]*)\}` pattern could not
    # match, so the `\frac` was never rewritten; it then fell through to the
    # generic "drop any leftover command" pass and the DIVISION WAS SILENTLY
    # DELETED. `PA = (Total - Breakdown) / Total` came out as
    # `PA = Total Hours - BreakdownTotal Hours`, which reads as a product and
    # states something the document does not. It hit 4 of 11 equations β€”
    # PA, PA_ij, UA_ij, PTY_ij.
    s = _unwrap_braced(s)

    # fold sub/superscripts, so `C _ { p }` becomes one token `C_p`. The content
    # is joined too: `_ { i j }` -> `_ij`, otherwise the space inside survives
    # and splits the token later ("_i j").
    s = re.sub(r"\s*_\s*\{\s*([^{}]*?)\s*\}", lambda m: "_" + _join_spaced(m.group(1)), s)
    s = re.sub(r"\s*\^\s*\{\s*([^{}]*?)\s*\}", lambda m: "^" + _join_spaced(m.group(1)), s)
    s = re.sub(r"\s*_\s*([A-Za-z0-9])", r"_\1", s)
    s = re.sub(r"\s*\^\s*([A-Za-z0-9])", r"^\1", s)

    s = _render_frac(s)

    s = re.sub(r"\\sqrt\s*\{([^{}]*)\}", r"sqrt(\1)", s)
    for k, v in _SYMBOLS.items():
        s = s.replace(k, f" {v} ")
    s = _WRAPPERS.sub(" ", s)      # leftover wrappers with no braces, e.g. `\mathrm x`
    s = _DROP_COMMANDS.sub(" ", s)
    s = re.sub(r"\\[A-Za-z]+", " ", s)     # any command left over
    s = s.replace("{", " ").replace("}", " ").replace("\\", " ")
    s = re.sub(r"[ \t\r\n]+", " ", s).strip()

    # now that spacing is uniform, rejoin the spelled-out words and numbers
    s = _join_spaced(s)

    s = s.replace(_WORD_GAP, " ")
    s = re.sub(r"\s+", " ", s).strip()

    # Last, once normal spacing is back: restore the word boundaries that the
    # LaTeX itself cannot tell us about.
    return restore_word_boundaries(s, vocabulary)


# --- HTML tables ---------------------------------------------------------

_CELL_END = re.compile(r"</\s*(?:td|th)\s*>", re.I)
_ROW_END = re.compile(r"</\s*tr\s*>", re.I)
_TAG = re.compile(r"<[^>]+>")

# Table cells routinely carry inline LaTeX ("$\frac{kVA}{...}$"). Stripping the
# HTML alone leaves that markup sitting in the prose, which is the same problem
# the formula rendering exists to solve β€” found in a real table on the first run.
_INLINE_MATH = re.compile(r"\$([^$]{1,400})\$")


def render_table(table_html: str, vocabulary: frozenset[str] | None = None) -> str:
    """HTML table -> plain rows, cells separated by ' | '.

    Keeps the reading order a person would use, so a term appearing only in a
    table header is still a mention.
    """
    s = _ROW_END.sub("\n", table_html)
    s = _CELL_END.sub(" | ", s)
    s = _TAG.sub(" ", s)
    s = html.unescape(s)
    s = _INLINE_MATH.sub(lambda m: " " + render_latex(m.group(1), vocabulary) + " ", s)

    rows = []
    for raw in s.split("\n"):
        cleaned = re.sub(r"[ \t]+", " ", raw).strip().strip("|").strip()
        if cleaned:
            rows.append(re.sub(r"\s*\|\s*", " | ", cleaned))
    return "\n".join(rows)