Download src/knowledge_extraction/validate/span_check.py from DataEyond/Agentic-Service-Data-Eyond-Catalog: direct link, hf CLI and curl.
- Browser
- Download file 19.9 kB
-
https://huggingface.co/spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/resolve/main/src/knowledge_extraction/validate/span_check.py
- Command line
-
hf download hf://spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/src/knowledge_extraction/validate/span_check.py
-
curl -L -o span_check.py https://huggingface.co/spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/resolve/main/src/knowledge_extraction/validate/span_check.py
19.9 kB
| """Verbatim span validation — the primary anti-hallucination control. | |
| Three rules that must not be relaxed: | |
| 1. **Normalise whitespace only.** Not case, not punctuation, not diacritics. | |
| Every additional normalisation is a hole a fabrication can fit through. | |
| 2. **On failure the FIELD becomes `None`** and the rejection is recorded, so a | |
| reviewer can see what the control caught rather than only what it let past. | |
| 3. **Never repair a failed span** by rewriting it to something that does match. | |
| A repaired span is an unfalsifiable claim, which is precisely what this | |
| control exists to prevent. | |
| If the provenance span itself is not verbatim, *every* guarded field on the | |
| entry is rejected: the entry's only link to evidence is broken, so nothing on it | |
| can be trusted. | |
| """ | |
| from __future__ import annotations | |
| import re | |
| import unicodedata | |
| from ..models import Branch, Chunk, RejectedField | |
| # Fields carrying a factual claim, and therefore span-guarded. | |
| GUARDED_FIELDS: dict[str, tuple[str, ...]] = { | |
| # v2 dropped `formula_latex` and `interpretation` from the glossary entry — | |
| # the first is now a reference to the formula branch, the second moved to | |
| # the Interpretation Pack. Neither is unguarded now; both are simply no | |
| # longer claimed here. | |
| "glossary": ("definition", "full_name", "source_wording"), | |
| "rule": ("statement", "condition", "consequence"), | |
| "formula": ("formula_latex",), | |
| # Was `()` until 2026-09-02: the branch asked the model to SUMMARISE, and a | |
| # summary is unverifiable by construction. It now asks the model to LOCATE, | |
| # and a located quote is checkable like any other. Both fields are | |
| # transcriptions, so both are also in `_TRANSCRIBED` below. | |
| "summary": ("title", "purpose_verbatim"), | |
| } | |
| def normalise_ws(text: str) -> str: | |
| return re.sub(r"\s+", " ", unicodedata.normalize("NFKC", text)).strip() | |
| def span_present(span: str, source: str) -> bool: | |
| if not span or not span.strip(): | |
| return False | |
| return normalise_ws(span) in normalise_ws(source) | |
| # Commands that change how a token is SET, never what it means. Unwrapped to | |
| # their contents, so `\mathrm{Total}` and `Total` compare equal. | |
| # | |
| # `mathrm`, `mathsf` and `text` are the three this corpus actually emits, across | |
| # 44 real strings (11 equations x 4 backends). The rest are kept because they | |
| # are the same class of command and cost nothing — but they are unobserved here, | |
| # so do not treat their presence as evidence of anything. | |
| _PRESENTATIONAL = ( | |
| "mathrm", "mathsf", "text", "mathbf", "mathit", "textrm", "operatorname" | |
| ) | |
| # Standalone style commands: no argument, no meaning. `\displaystyle` is the one | |
| # this corpus emits (pipeline, equation 4). | |
| # | |
| # The boundary is `(?![A-Za-z])`, NOT `\b`. `\b` does not match between `s` and | |
| # `_`, because `_` is a word character — so `\limits_{i}` kept its `\limits` | |
| # while `\limits x` lost it. Exactly the bug X22 found in the eval scorer, in a | |
| # different file. Anywhere a LaTeX command can be followed by a subscript, `\b` | |
| # is the wrong boundary. | |
| _BARE_STYLE = re.compile(r"\\displaystyle(?![A-Za-z])|\\limits(?![A-Za-z])|\\!") | |
| # Spacing, in every form LaTeX spells it. All equivalent to nothing here. | |
| # The negative lookbehind matters: without it the `\ ` rule eats one half of a | |
| # `\\` row separator that happens to be followed by a space, conflating a | |
| # structural marker with a cosmetic one. | |
| _SPACING = re.compile(r"~|\\[,;:]|\\quad\b|\\qquad\b|(?<!\\)\\(?=\s)") | |
| # `\begin{array}{lcr}` — the alignment spec is presentational; the same array | |
| # arrives as `{l}` from one backend and `{r}` from another. | |
| _ARRAY_ALIGN = re.compile(r"(\\begin\{array\})\{[lcr|@\s]*\}") | |
| # A row separator immediately before `\end{array}` is a trailing empty row. | |
| # Separators BETWEEN rows are structural and are left alone. | |
| _TRAILING_ROW = re.compile(r"(?:\\\\)+(?=\\end\{array\})") | |
| # `\left(` / ` | |
| # `\left(` / `\right)` size a delimiter; they do not change what it groups. | |
| _SIZING = re.compile(r"\\left(?=[([{|.])|\\right(?=[)\\]}|.])") | |
| # Every spelling of a fraction that takes two braced arguments. `\dfrac` and | |
| # `\tfrac` differ from `\frac` only in display size. Plain-TeX `{a \over b}` is | |
| # NOT handled: it is infix, so it needs a different parse, and MinerU has not | |
| # been observed emitting it. Unhandled means a false REJECTION, never a false | |
| # acceptance — the fail-safe direction. | |
| _FRACTIONS = ("\\frac{", "\\dfrac{", "\\tfrac{") | |
| # ⚠️ CORPUS-DERIVED ASSUMPTION, not a fact about LaTeX. | |
| # | |
| # In LaTeX generally, `\mathrm{x}` is a roman-set VARIABLE named x. Here it is | |
| # read as a multiplication sign, because that is how MinerU transcribes the `×` | |
| # glyph in this corpus: the real PA formula ends `\mathrm{x}100\%` under one | |
| # backend and `\mathsf{x}100\%` under another. | |
| # | |
| # Why it is tolerable: the rewrite is applied to BOTH sides of the comparison, | |
| # so a document that really does have a variable `x` set in roman converts it | |
| # identically in the claim and in the source, and they still match. The residual | |
| # risk is a contrived false acceptance (`Ma\times Capacity` matching a source | |
| # reading `MaxCapacity`), not a false rejection. | |
| # | |
| # Revisit this first if the pipeline ever meets a corpus where `x` is a genuine | |
| # variable name — algebra or statistics rather than mining parameters. | |
| _WRAPPED_TIMES = re.compile( | |
| r"\\(?:" + "|".join(_PRESENTATIONAL) + r")\{[xX]\}" | |
| ) | |
| def _unwrap_command(text: str, name: str) -> str: | |
| """Replace every `\\name{...}` with its contents, braces consumed. | |
| A balanced scanner rather than `\\{([^{}]*)\\}`, for the reason recorded in | |
| calibration §12: a regex character class cannot match a nested argument and | |
| fails SILENTLY, which is how `\\frac` lost its division there. The same | |
| mistake is available here and would have the same shape. | |
| """ | |
| token = "\\" + name | |
| out, i = [], 0 | |
| while True: | |
| start = text.find(token + "{", i) | |
| if start == -1: | |
| out.append(text[i:]) | |
| return "".join(out) | |
| out.append(text[i:start]) | |
| depth, j = 0, start + len(token) | |
| for j in range(start + len(token), len(text)): | |
| if text[j] == "{": | |
| depth += 1 | |
| elif text[j] == "}": | |
| depth -= 1 | |
| if depth == 0: | |
| break | |
| else: | |
| # Unbalanced: leave the remainder untouched rather than guess. | |
| out.append(text[start:]) | |
| return "".join(out) | |
| out.append(_unwrap_command(text[start + len(token) + 1 : j], name)) | |
| i = j + 1 | |
| def _collapse_redundant_parens(text: str) -> str: | |
| """`((x))` -> `(x)`, where the outer pair wraps nothing but the inner pair. | |
| Redundant grouping only. A pair that encloses anything else keeps both, | |
| because `(a+b)/c` and `a+(b/c)` are different formulas and the parentheses | |
| are the only thing saying so. | |
| """ | |
| while True: | |
| out, changed, i = [], False, 0 | |
| while i < len(text): | |
| if text[i] != "(": | |
| out.append(text[i]) | |
| i += 1 | |
| continue | |
| depth, j = 0, i | |
| for j in range(i, len(text)): | |
| if text[j] == "(": | |
| depth += 1 | |
| elif text[j] == ")": | |
| depth -= 1 | |
| if depth == 0: | |
| break | |
| else: | |
| out.append(text[i:]) | |
| i = len(text) | |
| break | |
| inner = text[i + 1 : j] | |
| if inner.startswith("(") and inner.endswith(")"): | |
| d, ok = 0, True | |
| for k, ch in enumerate(inner): | |
| d += (ch == "(") - (ch == ")") | |
| if d == 0 and k != len(inner) - 1: | |
| ok = False | |
| break | |
| if ok: # the outer pair wraps exactly one group | |
| out.append(inner) | |
| i = j + 1 | |
| changed = True | |
| continue | |
| out.append("(" + inner + ")") | |
| i = j + 1 | |
| text = "".join(out) | |
| if not changed: | |
| return text | |
| def _rewrite_fractions(text: str) -> str: | |
| """`\\frac{A}{B}` -> `(A)/(B)`, recursively, balanced. | |
| Canonicalised to the *prose* spelling of division on purpose. The model | |
| reads rendered prose — the document writes `Qty = (INPR Hours) / (Total | |
| Hours)` — so it answers with a slash while the markup carries `\\frac`. | |
| Both mean the same thing, and rejecting one for not being the other is the | |
| identity trap this whole function exists to avoid. | |
| The explicit parentheses are what keep it honest. Dropping them would make | |
| `\\frac{a+b}{c}` and `a+\\frac{b}{c}` normalise alike, which is a guard | |
| that cannot tell two different formulas apart. And X21 is still caught: | |
| its corruption leaves NO division operator at all, not a differently | |
| spelled one. | |
| """ | |
| hits = [(text.find(tok), tok) for tok in _FRACTIONS] | |
| hits = [(at, tok) for at, tok in hits if at != -1] | |
| if not hits: | |
| return text | |
| idx, token = min(hits) | |
| args, i = [], idx + len(token) - 1 # -1: the token carries its `{` | |
| skip = idx + len(token) - 1 | |
| for _ in range(2): | |
| if i >= len(text) or text[i] != "{": | |
| return text[:skip] + _rewrite_fractions(text[skip:]) | |
| depth = 0 | |
| for j in range(i, len(text)): | |
| if text[j] == "{": | |
| depth += 1 | |
| elif text[j] == "}": | |
| depth -= 1 | |
| if depth == 0: | |
| break | |
| else: | |
| return text | |
| args.append(text[i + 1 : j]) | |
| i = j + 1 | |
| numerator, denominator = (_rewrite_fractions(a) for a in args) | |
| body = f"({numerator})/({denominator})" | |
| return text[:idx] + body + _rewrite_fractions(text[i:]) | |
| def normalise_latex(text: str) -> str: | |
| """Canonical form for comparing a LaTeX claim against source markup. | |
| A different notion of equality from `normalise_ws`, not a weaker one. The | |
| prompt asks the model to *transcribe into* LaTeX — a translation — so | |
| testing string identity rejects correct answers written in a different but | |
| equivalent notation. Every rule below erases a difference that carries no | |
| meaning in LaTeX, and none erases structure. | |
| Never applied to prose. In prose whitespace is a word boundary and deleting | |
| it WOULD open a hole: "no data" would match "nodata". | |
| Each rule is grounded in markup this corpus actually emits (calibration §12 | |
| records the strings): MinerU spaces every character, wraps tokens in | |
| `\\mathrm{}`, writes `~` for a space, and — in the `pipeline` backend — | |
| writes multiplication as `\\mathrm{x}` or as a bare letter `x`. | |
| **Two known over-normalisations**, both measured rather than suspected, both | |
| accepted for this corpus and neither safe to assume elsewhere: | |
| * **Whitespace inside `\\text{}` is deleted along with the rest.** In text | |
| mode a space is a real word boundary, so `\\text{Total Hours}` and | |
| `\\text{TotalHours}` compare equal here. Two labels differing only by | |
| internal spacing therefore collide. Tolerated because a fabrication that | |
| differs from the truth by nothing but a space is not a threat worth the | |
| complexity of a mode-aware scanner — but it IS a loss of discrimination, | |
| not a neutral simplification. | |
| * **`\\mathrm{x}` is read as multiplication** — see `_WRAPPED_TIMES` for why | |
| that is a fact about this corpus rather than about LaTeX. | |
| What is deliberately NOT normalised, so the guard keeps its teeth: | |
| structure (`\\frac` is division and its absence is a different formula), | |
| precedence (the parentheses `\\frac` expands into), operand order, sub- vs | |
| superscript, `+` vs `-`, and `\\cdot` vs `\\times`. | |
| """ | |
| out = unicodedata.normalize("NFKC", text) | |
| # Explicit spacing first, while the whitespace it is defined against still | |
| # exists: `\ ` is a backslash followed by a space, and is unrecognisable | |
| # once the spaces are gone. | |
| out = _SPACING.sub("", out) | |
| # Then all remaining whitespace, BEFORE any structural rewrite. MinerU | |
| # writes `\mathrm { P A }` and `\frac { a } { b }` — spaced between the | |
| # command and its brace — so a scanner looking for `\mathrm{` finds | |
| # nothing until this has run. Getting this order wrong does not raise; it | |
| # silently skips every rewrite and compares raw markup. | |
| out = re.sub(r"\s+", "", out) | |
| # A wrapped roman `x` is this document's multiplication sign, not a | |
| # variable: the real PA formula ends `\mathrm{x}100\%` under one backend and | |
| # `\mathsf{x}100\%` under another. Must precede the unwrap, which would | |
| # otherwise destroy the distinction. | |
| out = _WRAPPED_TIMES.sub(r"\\times", out) | |
| out = _BARE_STYLE.sub("", out) | |
| out = _ARRAY_ALIGN.sub(r"\1", out) | |
| out = _TRAILING_ROW.sub("", out) | |
| for name in _PRESENTATIONAL: | |
| out = _unwrap_command(out, name) | |
| # `\cdot` is deliberately NOT folded into `\times`. It was, speculatively — | |
| # it appears nowhere in the 44 real strings — and it is wrong in general: on | |
| # vectors they are different operations, dot product versus cross product. | |
| # A guard that cannot tell those apart is worse than one that rejects a | |
| # `\cdot` it has never actually seen. | |
| out = out.replace("\\%", "%").replace("$", "") | |
| out = _SIZING.sub("", out) | |
| out = _rewrite_fractions(out) | |
| out = out.replace("{", "").replace("}", "") | |
| return _collapse_redundant_parens(out) | |
| # Every way this corpus writes multiplication once normalised. | |
| _MULTIPLICATION = re.compile(r"\\times") | |
| def multiplication_is_recoverable(source: str) -> bool: | |
| """Whether the source distinguishes a multiplication sign at all. | |
| The `pipeline` backend transcribes `×` as a bare letter `x`, with no | |
| `\\times` and no `\\mathrm` to mark it — calibration §12 records this as an | |
| open question, because `x` is also a plausible variable name. So on such an | |
| artifact the operator is genuinely absent from the source, and a claim that | |
| spells it out cannot be checked either way. | |
| Reported rather than guessed: resolving `x` to multiplication here would | |
| silently settle a decision that is deliberately still open, and would let | |
| `Ma\\times Capacity` match a source reading `MaxCapacity`. | |
| """ | |
| return bool(_MULTIPLICATION.search(normalise_latex(source))) | |
| def latex_present(span: str, source: str) -> bool: | |
| if not span or not span.strip(): | |
| return False | |
| return normalise_latex(span) in normalise_latex(source) | |
| def latex_source(chunks: list[Chunk]) -> str: | |
| """The markup a `formula_latex` claim is validated against.""" | |
| return "\n".join(fragment for chunk in chunks for fragment in chunk.latex) | |
| def evidence_text(chunk_ids: list[str], chunks: list[Chunk]) -> str: | |
| """The text a span is validated against. | |
| Includes each chunk's HEADING as well as its body. The heading is part of | |
| the source document and is often where a term is formally named — the | |
| reference standard heads a section "Physical of Availability (PA)" while | |
| the body never repeats the phrase. Excluding it would reject a correct, | |
| verbatim quotation of the document's own section title, which is precisely | |
| the wording we are required to preserve. | |
| """ | |
| by_id = {c.chunk_id: c for c in chunks} | |
| parts: list[str] = [] | |
| for chunk_id in chunk_ids: | |
| chunk = by_id.get(chunk_id) | |
| if chunk is None: | |
| continue | |
| if chunk.heading: | |
| parts.append(chunk.heading) | |
| parts.append(chunk.text) | |
| return "\n".join(parts) | |
| def _mark_latex(entry, verdict: str) -> None: | |
| """Record which guarantee the stored `formula_latex` carries. | |
| A never-throw seam like the rest of validation: an entry type without the | |
| field is left alone rather than raising, so a branch that gains a LaTeX | |
| field later opts in by declaring it. | |
| """ | |
| if hasattr(entry, "latex_verification"): | |
| entry.latex_verification = verdict | |
| def validate_entry( | |
| entry, branch: Branch, source_text: str, label: str, latex_text: str = "" | |
| ) -> tuple[object, list[RejectedField]]: | |
| """Returns `(entry, rejections)`; the entry is mutated in place. | |
| Also checks each guarded field's own value against the source where the | |
| field is expected to be quoted: `source_wording` and `full_name` are | |
| literal transcriptions, so a value that cannot be located is a silent | |
| normalisation — exactly the failure this pipeline is required to surface. | |
| """ | |
| rejections: list[RejectedField] = [] | |
| guarded = GUARDED_FIELDS.get(branch, ()) | |
| span_ok = span_present(entry.provenance.span, source_text) | |
| for field in guarded: | |
| value = getattr(entry, field, None) | |
| if value is None: | |
| continue | |
| if not span_ok: | |
| reason = "provenance.span not found verbatim in evidence" | |
| elif field in _TRANSCRIBED and not span_present(str(value), source_text): | |
| reason = f"{field} is not a verbatim transcription of the source" | |
| elif field in _LATEX_TRANSCRIBED and not latex_text.strip(): | |
| # No markup to compare against. Guarded by provenance alone, and | |
| # said so on the entry rather than left to look like the strong | |
| # guarantee. | |
| _mark_latex(entry, "unverified_no_markup") | |
| continue | |
| elif field in _LATEX_TRANSCRIBED: | |
| # Checked against the SOURCE MARKUP, not `text`. Only reachable when | |
| # the artifact actually carried markup for these chunks: an artifact | |
| # written before `Chunk.latex` crossed the seam has nothing to prove | |
| # the claim against, and "cannot verify" must not present as "failed | |
| # verification" — that would null every formula entry on every older | |
| # artifact. In that case the field stays guarded by provenance alone, | |
| # exactly as it was before this check existed. | |
| if latex_present(str(value), latex_text): | |
| _mark_latex(entry, "verified") | |
| continue | |
| # Same distinction one level down. A claim that spells out a | |
| # multiplication sign the source never distinguishes is unprovable | |
| # rather than wrong — `pipeline` writes `MOHH x Qty x PA` with no | |
| # operator to compare against. Rejecting here would null every | |
| # product formula on every `pipeline` artifact, silently, and it | |
| # would read as a bad model. | |
| if _MULTIPLICATION.search(normalise_latex(str(value))) and ( | |
| not multiplication_is_recoverable(latex_text) | |
| ): | |
| _mark_latex(entry, "unverified_operator") | |
| continue | |
| reason = f"{field} is not a transcription of the source markup" | |
| else: | |
| continue | |
| rejections.append( | |
| RejectedField( | |
| entry_term=label, | |
| field=field, | |
| offending_value=str(value)[:300], | |
| reason=reason, | |
| branch=branch, | |
| ) | |
| ) | |
| setattr(entry, field, None) | |
| return entry, rejections | |
| # Fields that claim to be copied from the document word for word. A definition | |
| # may legitimately be assembled across sentences; a "full name" may not. | |
| _TRANSCRIBED = frozenset({"full_name", "source_wording", "title", "purpose_verbatim"}) | |
| # Fields that claim to transcribe the document's MARKUP. Checked against | |
| # `Chunk.latex` with `latex_present`, never against `text` — putting one of | |
| # these in `_TRANSCRIBED` instead would reject every formula entry, because | |
| # `text` holds rendered prose and can never contain LaTeX. | |
| _LATEX_TRANSCRIBED = frozenset({"formula_latex"}) | |