Download src/knowledge_parsing/checks.py from DataEyond/Agentic-Service-Data-Eyond-Catalog: direct link, hf CLI and curl.
- Browser
- Download file 8.98 kB
-
https://huggingface.co/spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/resolve/main/src/knowledge_parsing/checks.py
- Command line
-
hf download hf://spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/src/knowledge_parsing/checks.py
-
curl -L -o checks.py https://huggingface.co/spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/resolve/main/src/knowledge_parsing/checks.py
8.98 kB
| """Quality checks on the parse — catching the damage that makes NO noise. | |
| The basis is a finding in `Hasil-Uji-MinerU.md`: the EOQ formula came out of the | |
| CPU pipeline with a digit changed, no error and no warning. Damage like that will | |
| never show up in a "finished" status — it has to be looked for. | |
| Everything here is a WARNING, never a failure: the document is still processed, | |
| but the warning is recorded in the manifest so quality cannot quietly degrade | |
| across hundreds of documents. | |
| """ | |
| from __future__ import annotations | |
| import re | |
| import unicodedata | |
| from collections import Counter | |
| from pathlib import Path | |
| from typing import Any | |
| # Digits inside latex. A technical formula almost always has a number or a divisor. | |
| _HAS_DIGIT = re.compile(r"\d") | |
| # Minimum length of a digit run before it is matched against the source. | |
| # | |
| # Measured on a real document: at a 2-digit threshold the genuine bug (55600) is | |
| # caught, but six false warnings come with it from short numbers — the PDF's text | |
| # layer is partly broken, so short numbers often fail to match despite being | |
| # correct. At 3 digits the bug is still caught and the noise disappears. | |
| MIN_DIGITS = 3 | |
| _ALL_DIGITS = re.compile(r"\d+") | |
| def _digit_runs(text: str, min_digits: int) -> set[str]: | |
| return {n for n in _ALL_DIGITS.findall(text) if len(n) >= min_digits} | |
| def _digits_from_latex(latex: str, min_digits: int = MIN_DIGITS) -> set[str]: | |
| """MinerU writes latex character-spaced ('5 5 6 0 0'), so spaces go first.""" | |
| return _digit_runs(re.sub(r"\s+", "", latex), min_digits) | |
| def _digits_per_page(pdf: Path, min_digits: int = MIN_DIGITS) -> dict[int, set[str]] | None: | |
| """Digit runs in the source PDF's own text layer, per page. | |
| Used as the reference. If the PDF cannot be read, return None and skip this | |
| check silently — not checking is better than warning falsely. | |
| """ | |
| try: | |
| from pypdf import PdfReader | |
| reader = PdfReader(str(pdf)) | |
| except Exception: | |
| return None | |
| result: dict[int, set[str]] = {} | |
| for i, page in enumerate(reader.pages): | |
| try: | |
| text = page.extract_text() or "" | |
| except Exception: | |
| text = "" | |
| # drop thousands separators & spaces so "$5,600" and "5 600" -> "5600" | |
| cleaned = re.sub(r"[, \s]", "", text) | |
| result[i] = _digit_runs(cleaned, min_digits) | |
| return result | |
| _WORD = re.compile(r"[A-Za-z][A-Za-z0-9]*") | |
| def vocabulary_per_page(pdf: Path) -> dict[int, frozenset[str]] | None: | |
| """Words in the source PDF's own text layer, per page. | |
| Used by `render.restore_word_boundaries` as the arbiter: MinerU writes | |
| formulas one character at a time and the word boundaries are lost there, | |
| while the PDF still holds them intact (`MOHH x Qty x PA`, not | |
| `MOHHxQtyxPAxUAxPty`). | |
| NFKC is required because formulas use a mathematical font — this page | |
| literally contains `𝑀𝑂𝐻𝐻` (U+1D440…) rather than ASCII, and without | |
| normalisation not a single word would match. | |
| Returns None if the PDF cannot be read; the caller then skips recovery | |
| silently, exactly as the digit check below does. | |
| """ | |
| try: | |
| from pypdf import PdfReader | |
| reader = PdfReader(str(pdf)) | |
| except Exception: | |
| return None | |
| result: dict[int, frozenset[str]] = {} | |
| for i, page in enumerate(reader.pages): | |
| try: | |
| text = page.extract_text() or "" | |
| except Exception: | |
| text = "" | |
| result[i] = frozenset(_WORD.findall(unicodedata.normalize("NFKC", text))) | |
| return result | |
| MIN_COVERAGE = 0.90 # share of MinerU's own words that must be found in the text layer | |
| MIN_WORDS_TO_JUDGE = 20 # below this there is no basis for a judgement | |
| def trustworthy_vocabulary( | |
| vocabulary: dict[int, frozenset[str]], | |
| items: list[dict[str, Any]], | |
| min_coverage: float = MIN_COVERAGE, | |
| ) -> dict[int, frozenset[str]]: | |
| """Drop pages whose text layer is not fit to arbitrate. | |
| `vocabulary_per_page` trusts the PDF text layer outright. That is safe when | |
| the layer is intact and dangerous when it is HALF there: a legitimate word | |
| starts looking unknown, and word-boundary recovery then splits it. | |
| The measure judges itself rather than relying on a constant tuned to one | |
| document: how many of MinerU's own prose words on that page are actually | |
| found in its text layer. A low score means the two read the same page | |
| differently — and a source that disagrees with the parser is not fit to | |
| arbitrate it. | |
| Measured on three real documents: 99.6% · 98.1% · 96.0%. Pages falling well | |
| below that lose their vocabulary, so recovery is skipped and the text is left | |
| as it is — damaged but honest, rather than repaired into something wrong. | |
| """ | |
| trusted: dict[int, frozenset[str]] = {} | |
| for page, words in vocabulary.items(): | |
| if not words: | |
| continue | |
| matched = total = 0 | |
| for x in items: | |
| if x.get("type") != "text" or x.get("page_idx") != page: | |
| continue | |
| for w in re.findall(r"[A-Za-z][A-Za-z0-9]*", x.get("text") or ""): | |
| if len(w) < 4: | |
| continue | |
| total += 1 | |
| matched += w in words | |
| # Too little prose to judge -> no basis for trusting it. | |
| if total < MIN_WORDS_TO_JUDGE or matched / total < min_coverage: | |
| continue | |
| trusted[page] = words | |
| return trusted | |
| def check_items( | |
| items: list[dict[str, Any]], | |
| min_latex_len: int = 8, | |
| source_pdf: Path | None = None, | |
| page_offset: int = 0, | |
| ) -> dict[str, Any]: | |
| """Return a summary plus the list of warnings for one document. | |
| `source_pdf` enables matching numbers against the original PDF — the only | |
| way to catch a number that changed silently (the EOQ bug: 5600 -> 55600). | |
| """ | |
| per_type = Counter(x.get("type") for x in items) | |
| pages = {x.get("page_idx") for x in items if x.get("page_idx") is not None} | |
| warnings: list[dict[str, Any]] = [] | |
| # 1. A page with no text at all -> it probably failed to read | |
| text_per_page = Counter( | |
| x.get("page_idx") for x in items | |
| if x.get("type") in {"text", "table", "equation"} and (x.get("text") or x.get("table_body")) | |
| ) | |
| for p in sorted(pages): | |
| if text_per_page.get(p, 0) == 0: | |
| warnings.append({"kind": "empty_page", "page": p}) | |
| # 2. Formula truncated / missing its numbers <- the EOQ bug | |
| for i, x in enumerate(items): | |
| if x.get("type") != "equation": | |
| continue | |
| latex = (x.get("text") or "").strip() | |
| if len(latex) < min_latex_len: | |
| warnings.append({"kind": "latex_too_short", "item": i, | |
| "page": x.get("page_idx"), "content": latex[:60]}) | |
| elif not _HAS_DIGIT.search(latex) and "\\frac" not in latex: | |
| warnings.append({"kind": "latex_without_digits", "item": i, | |
| "page": x.get("page_idx"), "content": latex[:60]}) | |
| # 3. ⭐ A number in a formula that is absent from the source PDF -> it changed | |
| # silently. This is the ONLY check that can catch the EOQ bug; every other | |
| # check in this file only inspects our own output, and a wrong number still | |
| # produces perfectly valid latex. | |
| source_digits = _digits_per_page(source_pdf) if source_pdf else None | |
| if source_digits is not None: | |
| for i, x in enumerate(items): | |
| if x.get("type") != "equation": | |
| continue | |
| p = x.get("page_idx", 0) + page_offset | |
| reference = source_digits.get(p) | |
| if not reference: # page with no text layer -> cannot be judged | |
| continue | |
| for number in sorted(_digits_from_latex(x.get("text") or "")): | |
| if number not in reference: | |
| warnings.append({ | |
| "kind": "digits_absent_from_source", "item": i, "page": p, | |
| "digits": number, | |
| "note": "a number in the formula was not found in the source PDF text", | |
| }) | |
| # 4. Table with no body | |
| for i, x in enumerate(items): | |
| if x.get("type") == "table" and not (x.get("table_body") or "").strip(): | |
| warnings.append({"kind": "empty_table", "item": i, "page": x.get("page_idx")}) | |
| # 5. Chart with no context whatsoever (the next stage can do nothing with it) | |
| for i, x in enumerate(items): | |
| if x.get("type") == "chart": | |
| has_any = (x.get("content") or "").strip() or (x.get("chart_caption") or []) | |
| if not has_any: | |
| warnings.append({"kind": "chart_without_context", "item": i, | |
| "page": x.get("page_idx")}) | |
| return { | |
| "items": len(items), | |
| "pages_detected": len(pages), | |
| "per_type": dict(per_type), | |
| "warnings": warnings, | |
| "warning_count": len(warnings), | |
| } | |