ishaq101's picture
/feat knowledge management (#20)
b68816f
Raw History Blame Contribute Delete
8.98 kB
"""Quality checks on the parse — catching the damage that makes NO noise.
The basis is a finding in `Hasil-Uji-MinerU.md`: the EOQ formula came out of the
CPU pipeline with a digit changed, no error and no warning. Damage like that will
never show up in a "finished" status — it has to be looked for.
Everything here is a WARNING, never a failure: the document is still processed,
but the warning is recorded in the manifest so quality cannot quietly degrade
across hundreds of documents.
"""
from __future__ import annotations
import re
import unicodedata
from collections import Counter
from pathlib import Path
from typing import Any
# Digits inside latex. A technical formula almost always has a number or a divisor.
_HAS_DIGIT = re.compile(r"\d")
# Minimum length of a digit run before it is matched against the source.
#
# Measured on a real document: at a 2-digit threshold the genuine bug (55600) is
# caught, but six false warnings come with it from short numbers — the PDF's text
# layer is partly broken, so short numbers often fail to match despite being
# correct. At 3 digits the bug is still caught and the noise disappears.
MIN_DIGITS = 3
_ALL_DIGITS = re.compile(r"\d+")
def _digit_runs(text: str, min_digits: int) -> set[str]:
return {n for n in _ALL_DIGITS.findall(text) if len(n) >= min_digits}
def _digits_from_latex(latex: str, min_digits: int = MIN_DIGITS) -> set[str]:
"""MinerU writes latex character-spaced ('5 5 6 0 0'), so spaces go first."""
return _digit_runs(re.sub(r"\s+", "", latex), min_digits)
def _digits_per_page(pdf: Path, min_digits: int = MIN_DIGITS) -> dict[int, set[str]] | None:
"""Digit runs in the source PDF's own text layer, per page.
Used as the reference. If the PDF cannot be read, return None and skip this
check silently — not checking is better than warning falsely.
"""
try:
from pypdf import PdfReader
reader = PdfReader(str(pdf))
except Exception:
return None
result: dict[int, set[str]] = {}
for i, page in enumerate(reader.pages):
try:
text = page.extract_text() or ""
except Exception:
text = ""
# drop thousands separators & spaces so "$5,600" and "5 600" -> "5600"
cleaned = re.sub(r"[, \s]", "", text)
result[i] = _digit_runs(cleaned, min_digits)
return result
_WORD = re.compile(r"[A-Za-z][A-Za-z0-9]*")
def vocabulary_per_page(pdf: Path) -> dict[int, frozenset[str]] | None:
"""Words in the source PDF's own text layer, per page.
Used by `render.restore_word_boundaries` as the arbiter: MinerU writes
formulas one character at a time and the word boundaries are lost there,
while the PDF still holds them intact (`MOHH x Qty x PA`, not
`MOHHxQtyxPAxUAxPty`).
NFKC is required because formulas use a mathematical font — this page
literally contains `𝑀𝑂𝐻𝐻` (U+1D440…) rather than ASCII, and without
normalisation not a single word would match.
Returns None if the PDF cannot be read; the caller then skips recovery
silently, exactly as the digit check below does.
"""
try:
from pypdf import PdfReader
reader = PdfReader(str(pdf))
except Exception:
return None
result: dict[int, frozenset[str]] = {}
for i, page in enumerate(reader.pages):
try:
text = page.extract_text() or ""
except Exception:
text = ""
result[i] = frozenset(_WORD.findall(unicodedata.normalize("NFKC", text)))
return result
MIN_COVERAGE = 0.90 # share of MinerU's own words that must be found in the text layer
MIN_WORDS_TO_JUDGE = 20 # below this there is no basis for a judgement
def trustworthy_vocabulary(
vocabulary: dict[int, frozenset[str]],
items: list[dict[str, Any]],
min_coverage: float = MIN_COVERAGE,
) -> dict[int, frozenset[str]]:
"""Drop pages whose text layer is not fit to arbitrate.
`vocabulary_per_page` trusts the PDF text layer outright. That is safe when
the layer is intact and dangerous when it is HALF there: a legitimate word
starts looking unknown, and word-boundary recovery then splits it.
The measure judges itself rather than relying on a constant tuned to one
document: how many of MinerU's own prose words on that page are actually
found in its text layer. A low score means the two read the same page
differently — and a source that disagrees with the parser is not fit to
arbitrate it.
Measured on three real documents: 99.6% · 98.1% · 96.0%. Pages falling well
below that lose their vocabulary, so recovery is skipped and the text is left
as it is — damaged but honest, rather than repaired into something wrong.
"""
trusted: dict[int, frozenset[str]] = {}
for page, words in vocabulary.items():
if not words:
continue
matched = total = 0
for x in items:
if x.get("type") != "text" or x.get("page_idx") != page:
continue
for w in re.findall(r"[A-Za-z][A-Za-z0-9]*", x.get("text") or ""):
if len(w) < 4:
continue
total += 1
matched += w in words
# Too little prose to judge -> no basis for trusting it.
if total < MIN_WORDS_TO_JUDGE or matched / total < min_coverage:
continue
trusted[page] = words
return trusted
def check_items(
items: list[dict[str, Any]],
min_latex_len: int = 8,
source_pdf: Path | None = None,
page_offset: int = 0,
) -> dict[str, Any]:
"""Return a summary plus the list of warnings for one document.
`source_pdf` enables matching numbers against the original PDF — the only
way to catch a number that changed silently (the EOQ bug: 5600 -> 55600).
"""
per_type = Counter(x.get("type") for x in items)
pages = {x.get("page_idx") for x in items if x.get("page_idx") is not None}
warnings: list[dict[str, Any]] = []
# 1. A page with no text at all -> it probably failed to read
text_per_page = Counter(
x.get("page_idx") for x in items
if x.get("type") in {"text", "table", "equation"} and (x.get("text") or x.get("table_body"))
)
for p in sorted(pages):
if text_per_page.get(p, 0) == 0:
warnings.append({"kind": "empty_page", "page": p})
# 2. Formula truncated / missing its numbers <- the EOQ bug
for i, x in enumerate(items):
if x.get("type") != "equation":
continue
latex = (x.get("text") or "").strip()
if len(latex) < min_latex_len:
warnings.append({"kind": "latex_too_short", "item": i,
"page": x.get("page_idx"), "content": latex[:60]})
elif not _HAS_DIGIT.search(latex) and "\\frac" not in latex:
warnings.append({"kind": "latex_without_digits", "item": i,
"page": x.get("page_idx"), "content": latex[:60]})
# 3. ⭐ A number in a formula that is absent from the source PDF -> it changed
# silently. This is the ONLY check that can catch the EOQ bug; every other
# check in this file only inspects our own output, and a wrong number still
# produces perfectly valid latex.
source_digits = _digits_per_page(source_pdf) if source_pdf else None
if source_digits is not None:
for i, x in enumerate(items):
if x.get("type") != "equation":
continue
p = x.get("page_idx", 0) + page_offset
reference = source_digits.get(p)
if not reference: # page with no text layer -> cannot be judged
continue
for number in sorted(_digits_from_latex(x.get("text") or "")):
if number not in reference:
warnings.append({
"kind": "digits_absent_from_source", "item": i, "page": p,
"digits": number,
"note": "a number in the formula was not found in the source PDF text",
})
# 4. Table with no body
for i, x in enumerate(items):
if x.get("type") == "table" and not (x.get("table_body") or "").strip():
warnings.append({"kind": "empty_table", "item": i, "page": x.get("page_idx")})
# 5. Chart with no context whatsoever (the next stage can do nothing with it)
for i, x in enumerate(items):
if x.get("type") == "chart":
has_any = (x.get("content") or "").strip() or (x.get("chart_caption") or [])
if not has_any:
warnings.append({"kind": "chart_without_context", "item": i,
"page": x.get("page_idx")})
return {
"items": len(items),
"pages_detected": len(pages),
"per_type": dict(per_type),
"warnings": warnings,
"warning_count": len(warnings),
}