Download src/knowledge_parsing/render.py from DataEyond/Agentic-Service-Data-Eyond-Catalog: direct link, hf CLI and curl.
- Browser
- Download file 12.1 kB
-
https://huggingface.co/spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/resolve/main/src/knowledge_parsing/render.py
- Command line
-
hf download hf://spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/src/knowledge_parsing/render.py
-
curl -L -o render.py https://huggingface.co/spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/resolve/main/src/knowledge_parsing/render.py
12.1 kB
| """Render MinerU's markup back into readable prose for `Chunk.text`. | |
| Why this exists — measured, not assumed. Running the same extraction pipeline, | |
| same document, same gold set, changing only the parse: | |
| PyMuPDF (plain text) recall 0.8537 35/41 | |
| MinerU, raw markup in text recall 0.7561 31/41 | |
| MinerU, formula + table rendered recall 0.8293 34/41 | |
| The term filter is an NER model reading prose. MinerU writes formulas with every | |
| character spaced out — `{ \\mathrm { P u r c h a s i n g ~ c o s t s } }` — and | |
| tables as HTML. Neither yields a single mention, even for terms the same model | |
| finds easily in plain text. | |
| The markup is not discarded: `Chunk.latex` and `Chunk.table_html` keep it | |
| verbatim, because the formula branch needs exactly that form. | |
| """ | |
| from __future__ import annotations | |
| import html | |
| import re | |
| # --- LaTeX --------------------------------------------------------------- | |
| _DROP_COMMANDS = re.compile( | |
| r"\\(?:qquad|quad|left|right|displaystyle|limits|nolimits)\b|\\[!,;:]" | |
| ) | |
| _WRAPPERS = re.compile(r"\\(?:mathrm|mathbf|mathit|mathsf|text|textrm|operatorname)\s*") | |
| # Operators map to non-letter glyphs on purpose. A letter here (e.g. "x" for | |
| # \times) would itself look like a spelled-out single character and get merged | |
| # into the token beside it — "C_p x T" becoming "C_p xT". | |
| _SYMBOLS = { | |
| r"\times": "×", r"\cdot": "·", r"\div": "/", r"\pm": "±", | |
| r"\leq": "≤", r"\geq": "≥", r"\neq": "≠", r"\approx": "≈", | |
| r"\%": "%", r"\$": "$", r"\&": "&", | |
| } | |
| # "P u r c h a s i n g" -> "Purchasing", "1 0 0" -> "100". Runs of >= 2 single | |
| # characters; letters and digits are joined separately so "C 5" is left alone. | |
| # Subscripts are folded first (below), so a real variable like `C_p` is already | |
| # one token and never gets swallowed into a neighbouring word. | |
| # | |
| # The boundaries are word-character classes, not `\S`. Two reasons, both found | |
| # on real equations: | |
| # - `\frac` rewriting INTRODUCES parentheses, and `(?<!\S)` refuses to match a | |
| # run touching one, stranding a letter at each end: | |
| # "(I N P R H o u r s)" -> "(I NPRHour s)". | |
| # - the trailing side lets `_`/`^` follow, so a run whose last letter carries a | |
| # subscript is not cut short: "Q t y_activity" -> "Qty_activity". | |
| _SPACED_RUN = re.compile( | |
| r"(?<![A-Za-z0-9_])(?:[A-Za-z]\s+){1,}[A-Za-z](?![A-Za-z0-9])" | |
| r"|(?<![A-Za-z0-9_])(?:\d\s+){1,}\d(?![A-Za-z0-9])" | |
| ) | |
| # Wrappers that take a braced argument. Handled by brace matching rather than | |
| # `_WRAPPERS` below, because merely deleting the command name leaves its braces | |
| # behind — and those braces are what stopped `\frac` from matching. | |
| _BRACED_WRAPPER = re.compile( | |
| r"\\(?:mathrm|mathbf|mathit|mathsf|mathtt|text|textrm|textbf|operatorname)\s*(?=\{)" | |
| ) | |
| # `~` is a non-breaking SPACE in LaTeX, i.e. a real word boundary. It has to | |
| # survive the whitespace pass, otherwise "c o s t s" after it merges into the | |
| # preceding word and "Purchasing costs" becomes "Purchasingcosts". | |
| _WORD_GAP = "\x00" | |
| def _join_spaced(text: str) -> str: | |
| """Join runs of characters spelled out one by one: 'i j' -> 'ij'.""" | |
| return _SPACED_RUN.sub(lambda m: m.group(0).replace(" ", ""), text) | |
| # Minimum token length eligible for splitting. Measured on real documents: 6, 4 | |
| # and 3 give identical results on equations, and 3 additionally recovers `MOHHx` | |
| # / `xPA_A` inside table cells. Dropping to 2 starts doing DAMAGE — `PA_ij` | |
| # splits into `PA_i j`, because `i` and `j` are both valid words on that page. | |
| MIN_TOKEN_TO_SPLIT = 3 | |
| _MAX_WORD_LEN = 24 # a sane single-word length; bounds the DP work | |
| _TOKEN = re.compile(r"[A-Za-z][A-Za-z0-9]*") | |
| def _segment(token: str, vocabulary: frozenset[str]) -> list[str] | None: | |
| """Split `token` into words that are ALL present in `vocabulary`. | |
| Returns None when no valid split exists, OR when more than one shortest | |
| split is equally valid. Ambiguity is left alone — this module does not | |
| guess, and a token that cannot be recovered with certainty is better left | |
| damaged than repaired into the wrong form with no trace. | |
| """ | |
| n = len(token) | |
| # dp[i] = (fewest words for token[:i], how many ways reach that, the parts) | |
| dp: list[tuple[int, int, list[str]] | None] = [None] * (n + 1) | |
| dp[0] = (0, 1, []) | |
| for i in range(1, n + 1): | |
| best: int | None = None | |
| ways = 0 | |
| parts: list[str] = [] | |
| for j in range(max(0, i - _MAX_WORD_LEN), i): | |
| prev = dp[j] | |
| if prev is None or token[j:i] not in vocabulary: | |
| continue | |
| prev_len, prev_ways, prev_words = prev | |
| candidate = prev_len + 1 | |
| if best is None or candidate < best: | |
| best, ways, parts = candidate, prev_ways, prev_words + [token[j:i]] | |
| elif candidate == best: | |
| ways += prev_ways | |
| if best is not None: | |
| dp[i] = (best, ways, parts) | |
| result = dp[n] | |
| if result is None: | |
| return None | |
| _, ways, parts = result | |
| if ways != 1 or len(parts) < 2: | |
| return None | |
| return parts | |
| def restore_word_boundaries(text: str, vocabulary: frozenset[str] | None) -> str: | |
| """Put back the spaces lost when a character-spaced run was joined up. | |
| MinerU writes formulas one character at a time (`T o t a l H o u r s`), and | |
| once those are joined the word boundaries cannot be recovered from the LaTeX | |
| itself: `Total Hours` and `TotalHours` become indistinguishable. A | |
| multiplication sign written as the letter `x` fuses in too, because it looks | |
| exactly like any other spaced character. | |
| The source PDF's text layer STILL holds those boundaries, so the answer is | |
| read from the document rather than guessed. A token already present as a word | |
| in the source is never touched — which means a document that writes its | |
| formulas correctly skips this path entirely. | |
| """ | |
| if not vocabulary: | |
| return text | |
| replacements: dict[str, str] = {} | |
| for token in set(_TOKEN.findall(text)): | |
| if len(token) < MIN_TOKEN_TO_SPLIT or token in vocabulary: | |
| continue | |
| parts = _segment(token, vocabulary) | |
| if parts: | |
| replacements[token] = " ".join(parts) | |
| for token, joined in sorted(replacements.items(), key=lambda kv: -len(kv[0])): | |
| text = re.sub(rf"(?<![A-Za-z0-9]){re.escape(token)}(?![A-Za-z0-9])", joined, text) | |
| return text | |
| def _braced_argument(s: str, i: int) -> tuple[str | None, int]: | |
| """The BALANCED brace contents starting at `s[i] == '{'`. | |
| A regex `\\{([^{}]*)\\}` cannot be used here: real arguments contain nested | |
| braces (`\\mathrm{...}`, `\\sum _{ij}`), and that pattern fails to match | |
| silently rather than raising. | |
| """ | |
| depth = 0 | |
| for j in range(i, len(s)): | |
| if s[j] == "{": | |
| depth += 1 | |
| elif s[j] == "}": | |
| depth -= 1 | |
| if depth == 0: | |
| return s[i + 1:j], j + 1 | |
| return None, i | |
| def _unwrap_braced(s: str) -> str: | |
| r"""`\mathrm{X}` -> `X`, consuming its braces along with it.""" | |
| for _ in range(8): | |
| m = _BRACED_WRAPPER.search(s) | |
| if not m: | |
| break | |
| inner, end = _braced_argument(s, m.end()) | |
| if inner is None: | |
| break | |
| s = s[:m.start()] + " " + inner + " " + s[end:] | |
| return s | |
| def _render_frac(s: str) -> str: | |
| r"""`\frac{a}{b}` -> `(a) / (b)`, brace-aware and recursive.""" | |
| for _ in range(8): | |
| m = re.search(r"\\d?frac\s*(?=\{)", s) | |
| if not m: | |
| break | |
| numerator, after = _braced_argument(s, s.index("{", m.end() - 1)) | |
| if numerator is None: | |
| break | |
| denom_start = s.find("{", after) | |
| if denom_start == -1: | |
| break | |
| denominator, end = _braced_argument(s, denom_start) | |
| if denominator is None: | |
| break | |
| s = (s[:m.start()] + " (" + _render_frac(numerator) + ") / (" | |
| + _render_frac(denominator) + ") " + s[end:]) | |
| return s | |
| def render_latex(latex: str, vocabulary: frozenset[str] | None = None) -> str: | |
| """LaTeX -> a readable line. Best effort: it feeds a term filter, not a parser. | |
| `vocabulary` is the source page's own word list, used only to put back word | |
| boundaries this function cannot recover on its own — see | |
| `restore_word_boundaries`. Omitting it reproduces the previous behaviour | |
| exactly, so existing callers are unaffected. | |
| """ | |
| s = latex.replace("$$", " ").replace("$", " ") | |
| s = s.replace("~", _WORD_GAP) | |
| # array/matrix scaffolding carries no meaning for a term filter | |
| s = re.sub(r"\\(?:begin|end)\s*\{[^{}]*\}", " ", s) | |
| s = re.sub(r"\{\s*(?:[rlc|]\s*){1,8}\}", " ", s) # column spec, e.g. { r l } | |
| s = s.replace("\\\\", " ; ").replace("&", " ") | |
| # ORDER MATTERS, and this is the ordering decision: unwrap the braced | |
| # wrappers BEFORE touching `\frac`. | |
| # | |
| # `\frac`'s arguments in real documents are not brace-free — the BUMA | |
| # standard writes `\frac { \mathrm { T o t a l } ~ H o u r s - ... } { ... }`. | |
| # With the wrappers still in place the old `\{([^{}]*)\}` pattern could not | |
| # match, so the `\frac` was never rewritten; it then fell through to the | |
| # generic "drop any leftover command" pass and the DIVISION WAS SILENTLY | |
| # DELETED. `PA = (Total - Breakdown) / Total` came out as | |
| # `PA = Total Hours - BreakdownTotal Hours`, which reads as a product and | |
| # states something the document does not. It hit 4 of 11 equations — | |
| # PA, PA_ij, UA_ij, PTY_ij. | |
| s = _unwrap_braced(s) | |
| # fold sub/superscripts, so `C _ { p }` becomes one token `C_p`. The content | |
| # is joined too: `_ { i j }` -> `_ij`, otherwise the space inside survives | |
| # and splits the token later ("_i j"). | |
| s = re.sub(r"\s*_\s*\{\s*([^{}]*?)\s*\}", lambda m: "_" + _join_spaced(m.group(1)), s) | |
| s = re.sub(r"\s*\^\s*\{\s*([^{}]*?)\s*\}", lambda m: "^" + _join_spaced(m.group(1)), s) | |
| s = re.sub(r"\s*_\s*([A-Za-z0-9])", r"_\1", s) | |
| s = re.sub(r"\s*\^\s*([A-Za-z0-9])", r"^\1", s) | |
| s = _render_frac(s) | |
| s = re.sub(r"\\sqrt\s*\{([^{}]*)\}", r"sqrt(\1)", s) | |
| for k, v in _SYMBOLS.items(): | |
| s = s.replace(k, f" {v} ") | |
| s = _WRAPPERS.sub(" ", s) # leftover wrappers with no braces, e.g. `\mathrm x` | |
| s = _DROP_COMMANDS.sub(" ", s) | |
| s = re.sub(r"\\[A-Za-z]+", " ", s) # any command left over | |
| s = s.replace("{", " ").replace("}", " ").replace("\\", " ") | |
| s = re.sub(r"[ \t\r\n]+", " ", s).strip() | |
| # now that spacing is uniform, rejoin the spelled-out words and numbers | |
| s = _join_spaced(s) | |
| s = s.replace(_WORD_GAP, " ") | |
| s = re.sub(r"\s+", " ", s).strip() | |
| # Last, once normal spacing is back: restore the word boundaries that the | |
| # LaTeX itself cannot tell us about. | |
| return restore_word_boundaries(s, vocabulary) | |
| # --- HTML tables --------------------------------------------------------- | |
| _CELL_END = re.compile(r"</\s*(?:td|th)\s*>", re.I) | |
| _ROW_END = re.compile(r"</\s*tr\s*>", re.I) | |
| _TAG = re.compile(r"<[^>]+>") | |
| # Table cells routinely carry inline LaTeX ("$\frac{kVA}{...}$"). Stripping the | |
| # HTML alone leaves that markup sitting in the prose, which is the same problem | |
| # the formula rendering exists to solve — found in a real table on the first run. | |
| _INLINE_MATH = re.compile(r"\$([^$]{1,400})\$") | |
| def render_table(table_html: str, vocabulary: frozenset[str] | None = None) -> str: | |
| """HTML table -> plain rows, cells separated by ' | '. | |
| Keeps the reading order a person would use, so a term appearing only in a | |
| table header is still a mention. | |
| """ | |
| s = _ROW_END.sub("\n", table_html) | |
| s = _CELL_END.sub(" | ", s) | |
| s = _TAG.sub(" ", s) | |
| s = html.unescape(s) | |
| s = _INLINE_MATH.sub(lambda m: " " + render_latex(m.group(1), vocabulary) + " ", s) | |
| rows = [] | |
| for raw in s.split("\n"): | |
| cleaned = re.sub(r"[ \t]+", " ", raw).strip().strip("|").strip() | |
| if cleaned: | |
| rows.append(re.sub(r"\s*\|\s*", " | ", cleaned)) | |
| return "\n".join(rows) | |