ishaq101's picture sofhiaazzhr's picture
/feat knowledge management (#20)
b68816f
Raw History Blame Contribute Delete
19.9 kB
"""Verbatim span validation — the primary anti-hallucination control.
Three rules that must not be relaxed:
1. **Normalise whitespace only.** Not case, not punctuation, not diacritics.
Every additional normalisation is a hole a fabrication can fit through.
2. **On failure the FIELD becomes `None`** and the rejection is recorded, so a
reviewer can see what the control caught rather than only what it let past.
3. **Never repair a failed span** by rewriting it to something that does match.
A repaired span is an unfalsifiable claim, which is precisely what this
control exists to prevent.
If the provenance span itself is not verbatim, *every* guarded field on the
entry is rejected: the entry's only link to evidence is broken, so nothing on it
can be trusted.
"""
from __future__ import annotations
import re
import unicodedata
from ..models import Branch, Chunk, RejectedField
# Fields carrying a factual claim, and therefore span-guarded.
GUARDED_FIELDS: dict[str, tuple[str, ...]] = {
# v2 dropped `formula_latex` and `interpretation` from the glossary entry —
# the first is now a reference to the formula branch, the second moved to
# the Interpretation Pack. Neither is unguarded now; both are simply no
# longer claimed here.
"glossary": ("definition", "full_name", "source_wording"),
"rule": ("statement", "condition", "consequence"),
"formula": ("formula_latex",),
# Was `()` until 2026-09-02: the branch asked the model to SUMMARISE, and a
# summary is unverifiable by construction. It now asks the model to LOCATE,
# and a located quote is checkable like any other. Both fields are
# transcriptions, so both are also in `_TRANSCRIBED` below.
"summary": ("title", "purpose_verbatim"),
}
def normalise_ws(text: str) -> str:
return re.sub(r"\s+", " ", unicodedata.normalize("NFKC", text)).strip()
def span_present(span: str, source: str) -> bool:
if not span or not span.strip():
return False
return normalise_ws(span) in normalise_ws(source)
# Commands that change how a token is SET, never what it means. Unwrapped to
# their contents, so `\mathrm{Total}` and `Total` compare equal.
#
# `mathrm`, `mathsf` and `text` are the three this corpus actually emits, across
# 44 real strings (11 equations x 4 backends). The rest are kept because they
# are the same class of command and cost nothing — but they are unobserved here,
# so do not treat their presence as evidence of anything.
_PRESENTATIONAL = (
"mathrm", "mathsf", "text", "mathbf", "mathit", "textrm", "operatorname"
)
# Standalone style commands: no argument, no meaning. `\displaystyle` is the one
# this corpus emits (pipeline, equation 4).
#
# The boundary is `(?![A-Za-z])`, NOT `\b`. `\b` does not match between `s` and
# `_`, because `_` is a word character — so `\limits_{i}` kept its `\limits`
# while `\limits x` lost it. Exactly the bug X22 found in the eval scorer, in a
# different file. Anywhere a LaTeX command can be followed by a subscript, `\b`
# is the wrong boundary.
_BARE_STYLE = re.compile(r"\\displaystyle(?![A-Za-z])|\\limits(?![A-Za-z])|\\!")
# Spacing, in every form LaTeX spells it. All equivalent to nothing here.
# The negative lookbehind matters: without it the `\ ` rule eats one half of a
# `\\` row separator that happens to be followed by a space, conflating a
# structural marker with a cosmetic one.
_SPACING = re.compile(r"~|\\[,;:]|\\quad\b|\\qquad\b|(?<!\\)\\(?=\s)")
# `\begin{array}{lcr}` — the alignment spec is presentational; the same array
# arrives as `{l}` from one backend and `{r}` from another.
_ARRAY_ALIGN = re.compile(r"(\\begin\{array\})\{[lcr|@\s]*\}")
# A row separator immediately before `\end{array}` is a trailing empty row.
# Separators BETWEEN rows are structural and are left alone.
_TRAILING_ROW = re.compile(r"(?:\\\\)+(?=\\end\{array\})")
# `\left(` / `
# `\left(` / `\right)` size a delimiter; they do not change what it groups.
_SIZING = re.compile(r"\\left(?=[([{|.])|\\right(?=[)\\]}|.])")
# Every spelling of a fraction that takes two braced arguments. `\dfrac` and
# `\tfrac` differ from `\frac` only in display size. Plain-TeX `{a \over b}` is
# NOT handled: it is infix, so it needs a different parse, and MinerU has not
# been observed emitting it. Unhandled means a false REJECTION, never a false
# acceptance — the fail-safe direction.
_FRACTIONS = ("\\frac{", "\\dfrac{", "\\tfrac{")
# ⚠️ CORPUS-DERIVED ASSUMPTION, not a fact about LaTeX.
#
# In LaTeX generally, `\mathrm{x}` is a roman-set VARIABLE named x. Here it is
# read as a multiplication sign, because that is how MinerU transcribes the `×`
# glyph in this corpus: the real PA formula ends `\mathrm{x}100\%` under one
# backend and `\mathsf{x}100\%` under another.
#
# Why it is tolerable: the rewrite is applied to BOTH sides of the comparison,
# so a document that really does have a variable `x` set in roman converts it
# identically in the claim and in the source, and they still match. The residual
# risk is a contrived false acceptance (`Ma\times Capacity` matching a source
# reading `MaxCapacity`), not a false rejection.
#
# Revisit this first if the pipeline ever meets a corpus where `x` is a genuine
# variable name — algebra or statistics rather than mining parameters.
_WRAPPED_TIMES = re.compile(
r"\\(?:" + "|".join(_PRESENTATIONAL) + r")\{[xX]\}"
)
def _unwrap_command(text: str, name: str) -> str:
"""Replace every `\\name{...}` with its contents, braces consumed.
A balanced scanner rather than `\\{([^{}]*)\\}`, for the reason recorded in
calibration §12: a regex character class cannot match a nested argument and
fails SILENTLY, which is how `\\frac` lost its division there. The same
mistake is available here and would have the same shape.
"""
token = "\\" + name
out, i = [], 0
while True:
start = text.find(token + "{", i)
if start == -1:
out.append(text[i:])
return "".join(out)
out.append(text[i:start])
depth, j = 0, start + len(token)
for j in range(start + len(token), len(text)):
if text[j] == "{":
depth += 1
elif text[j] == "}":
depth -= 1
if depth == 0:
break
else:
# Unbalanced: leave the remainder untouched rather than guess.
out.append(text[start:])
return "".join(out)
out.append(_unwrap_command(text[start + len(token) + 1 : j], name))
i = j + 1
def _collapse_redundant_parens(text: str) -> str:
"""`((x))` -> `(x)`, where the outer pair wraps nothing but the inner pair.
Redundant grouping only. A pair that encloses anything else keeps both,
because `(a+b)/c` and `a+(b/c)` are different formulas and the parentheses
are the only thing saying so.
"""
while True:
out, changed, i = [], False, 0
while i < len(text):
if text[i] != "(":
out.append(text[i])
i += 1
continue
depth, j = 0, i
for j in range(i, len(text)):
if text[j] == "(":
depth += 1
elif text[j] == ")":
depth -= 1
if depth == 0:
break
else:
out.append(text[i:])
i = len(text)
break
inner = text[i + 1 : j]
if inner.startswith("(") and inner.endswith(")"):
d, ok = 0, True
for k, ch in enumerate(inner):
d += (ch == "(") - (ch == ")")
if d == 0 and k != len(inner) - 1:
ok = False
break
if ok: # the outer pair wraps exactly one group
out.append(inner)
i = j + 1
changed = True
continue
out.append("(" + inner + ")")
i = j + 1
text = "".join(out)
if not changed:
return text
def _rewrite_fractions(text: str) -> str:
"""`\\frac{A}{B}` -> `(A)/(B)`, recursively, balanced.
Canonicalised to the *prose* spelling of division on purpose. The model
reads rendered prose — the document writes `Qty = (INPR Hours) / (Total
Hours)` — so it answers with a slash while the markup carries `\\frac`.
Both mean the same thing, and rejecting one for not being the other is the
identity trap this whole function exists to avoid.
The explicit parentheses are what keep it honest. Dropping them would make
`\\frac{a+b}{c}` and `a+\\frac{b}{c}` normalise alike, which is a guard
that cannot tell two different formulas apart. And X21 is still caught:
its corruption leaves NO division operator at all, not a differently
spelled one.
"""
hits = [(text.find(tok), tok) for tok in _FRACTIONS]
hits = [(at, tok) for at, tok in hits if at != -1]
if not hits:
return text
idx, token = min(hits)
args, i = [], idx + len(token) - 1 # -1: the token carries its `{`
skip = idx + len(token) - 1
for _ in range(2):
if i >= len(text) or text[i] != "{":
return text[:skip] + _rewrite_fractions(text[skip:])
depth = 0
for j in range(i, len(text)):
if text[j] == "{":
depth += 1
elif text[j] == "}":
depth -= 1
if depth == 0:
break
else:
return text
args.append(text[i + 1 : j])
i = j + 1
numerator, denominator = (_rewrite_fractions(a) for a in args)
body = f"({numerator})/({denominator})"
return text[:idx] + body + _rewrite_fractions(text[i:])
def normalise_latex(text: str) -> str:
"""Canonical form for comparing a LaTeX claim against source markup.
A different notion of equality from `normalise_ws`, not a weaker one. The
prompt asks the model to *transcribe into* LaTeX — a translation — so
testing string identity rejects correct answers written in a different but
equivalent notation. Every rule below erases a difference that carries no
meaning in LaTeX, and none erases structure.
Never applied to prose. In prose whitespace is a word boundary and deleting
it WOULD open a hole: "no data" would match "nodata".
Each rule is grounded in markup this corpus actually emits (calibration §12
records the strings): MinerU spaces every character, wraps tokens in
`\\mathrm{}`, writes `~` for a space, and — in the `pipeline` backend —
writes multiplication as `\\mathrm{x}` or as a bare letter `x`.
**Two known over-normalisations**, both measured rather than suspected, both
accepted for this corpus and neither safe to assume elsewhere:
* **Whitespace inside `\\text{}` is deleted along with the rest.** In text
mode a space is a real word boundary, so `\\text{Total Hours}` and
`\\text{TotalHours}` compare equal here. Two labels differing only by
internal spacing therefore collide. Tolerated because a fabrication that
differs from the truth by nothing but a space is not a threat worth the
complexity of a mode-aware scanner — but it IS a loss of discrimination,
not a neutral simplification.
* **`\\mathrm{x}` is read as multiplication** — see `_WRAPPED_TIMES` for why
that is a fact about this corpus rather than about LaTeX.
What is deliberately NOT normalised, so the guard keeps its teeth:
structure (`\\frac` is division and its absence is a different formula),
precedence (the parentheses `\\frac` expands into), operand order, sub- vs
superscript, `+` vs `-`, and `\\cdot` vs `\\times`.
"""
out = unicodedata.normalize("NFKC", text)
# Explicit spacing first, while the whitespace it is defined against still
# exists: `\ ` is a backslash followed by a space, and is unrecognisable
# once the spaces are gone.
out = _SPACING.sub("", out)
# Then all remaining whitespace, BEFORE any structural rewrite. MinerU
# writes `\mathrm { P A }` and `\frac { a } { b }` — spaced between the
# command and its brace — so a scanner looking for `\mathrm{` finds
# nothing until this has run. Getting this order wrong does not raise; it
# silently skips every rewrite and compares raw markup.
out = re.sub(r"\s+", "", out)
# A wrapped roman `x` is this document's multiplication sign, not a
# variable: the real PA formula ends `\mathrm{x}100\%` under one backend and
# `\mathsf{x}100\%` under another. Must precede the unwrap, which would
# otherwise destroy the distinction.
out = _WRAPPED_TIMES.sub(r"\\times", out)
out = _BARE_STYLE.sub("", out)
out = _ARRAY_ALIGN.sub(r"\1", out)
out = _TRAILING_ROW.sub("", out)
for name in _PRESENTATIONAL:
out = _unwrap_command(out, name)
# `\cdot` is deliberately NOT folded into `\times`. It was, speculatively —
# it appears nowhere in the 44 real strings — and it is wrong in general: on
# vectors they are different operations, dot product versus cross product.
# A guard that cannot tell those apart is worse than one that rejects a
# `\cdot` it has never actually seen.
out = out.replace("\\%", "%").replace("$", "")
out = _SIZING.sub("", out)
out = _rewrite_fractions(out)
out = out.replace("{", "").replace("}", "")
return _collapse_redundant_parens(out)
# Every way this corpus writes multiplication once normalised.
_MULTIPLICATION = re.compile(r"\\times")
def multiplication_is_recoverable(source: str) -> bool:
"""Whether the source distinguishes a multiplication sign at all.
The `pipeline` backend transcribes `×` as a bare letter `x`, with no
`\\times` and no `\\mathrm` to mark it — calibration §12 records this as an
open question, because `x` is also a plausible variable name. So on such an
artifact the operator is genuinely absent from the source, and a claim that
spells it out cannot be checked either way.
Reported rather than guessed: resolving `x` to multiplication here would
silently settle a decision that is deliberately still open, and would let
`Ma\\times Capacity` match a source reading `MaxCapacity`.
"""
return bool(_MULTIPLICATION.search(normalise_latex(source)))
def latex_present(span: str, source: str) -> bool:
if not span or not span.strip():
return False
return normalise_latex(span) in normalise_latex(source)
def latex_source(chunks: list[Chunk]) -> str:
"""The markup a `formula_latex` claim is validated against."""
return "\n".join(fragment for chunk in chunks for fragment in chunk.latex)
def evidence_text(chunk_ids: list[str], chunks: list[Chunk]) -> str:
"""The text a span is validated against.
Includes each chunk's HEADING as well as its body. The heading is part of
the source document and is often where a term is formally named — the
reference standard heads a section "Physical of Availability (PA)" while
the body never repeats the phrase. Excluding it would reject a correct,
verbatim quotation of the document's own section title, which is precisely
the wording we are required to preserve.
"""
by_id = {c.chunk_id: c for c in chunks}
parts: list[str] = []
for chunk_id in chunk_ids:
chunk = by_id.get(chunk_id)
if chunk is None:
continue
if chunk.heading:
parts.append(chunk.heading)
parts.append(chunk.text)
return "\n".join(parts)
def _mark_latex(entry, verdict: str) -> None:
"""Record which guarantee the stored `formula_latex` carries.
A never-throw seam like the rest of validation: an entry type without the
field is left alone rather than raising, so a branch that gains a LaTeX
field later opts in by declaring it.
"""
if hasattr(entry, "latex_verification"):
entry.latex_verification = verdict
def validate_entry(
entry, branch: Branch, source_text: str, label: str, latex_text: str = ""
) -> tuple[object, list[RejectedField]]:
"""Returns `(entry, rejections)`; the entry is mutated in place.
Also checks each guarded field's own value against the source where the
field is expected to be quoted: `source_wording` and `full_name` are
literal transcriptions, so a value that cannot be located is a silent
normalisation — exactly the failure this pipeline is required to surface.
"""
rejections: list[RejectedField] = []
guarded = GUARDED_FIELDS.get(branch, ())
span_ok = span_present(entry.provenance.span, source_text)
for field in guarded:
value = getattr(entry, field, None)
if value is None:
continue
if not span_ok:
reason = "provenance.span not found verbatim in evidence"
elif field in _TRANSCRIBED and not span_present(str(value), source_text):
reason = f"{field} is not a verbatim transcription of the source"
elif field in _LATEX_TRANSCRIBED and not latex_text.strip():
# No markup to compare against. Guarded by provenance alone, and
# said so on the entry rather than left to look like the strong
# guarantee.
_mark_latex(entry, "unverified_no_markup")
continue
elif field in _LATEX_TRANSCRIBED:
# Checked against the SOURCE MARKUP, not `text`. Only reachable when
# the artifact actually carried markup for these chunks: an artifact
# written before `Chunk.latex` crossed the seam has nothing to prove
# the claim against, and "cannot verify" must not present as "failed
# verification" — that would null every formula entry on every older
# artifact. In that case the field stays guarded by provenance alone,
# exactly as it was before this check existed.
if latex_present(str(value), latex_text):
_mark_latex(entry, "verified")
continue
# Same distinction one level down. A claim that spells out a
# multiplication sign the source never distinguishes is unprovable
# rather than wrong — `pipeline` writes `MOHH x Qty x PA` with no
# operator to compare against. Rejecting here would null every
# product formula on every `pipeline` artifact, silently, and it
# would read as a bad model.
if _MULTIPLICATION.search(normalise_latex(str(value))) and (
not multiplication_is_recoverable(latex_text)
):
_mark_latex(entry, "unverified_operator")
continue
reason = f"{field} is not a transcription of the source markup"
else:
continue
rejections.append(
RejectedField(
entry_term=label,
field=field,
offending_value=str(value)[:300],
reason=reason,
branch=branch,
)
)
setattr(entry, field, None)
return entry, rejections
# Fields that claim to be copied from the document word for word. A definition
# may legitimately be assembled across sentences; a "full name" may not.
_TRANSCRIBED = frozenset({"full_name", "source_wording", "title", "purpose_verbatim"})
# Fields that claim to transcribe the document's MARKUP. Checked against
# `Chunk.latex` with `latex_present`, never against `text` — putting one of
# these in `_TRANSCRIBED` instead would reject every formula entry, because
# `text` holds rendered prose and can never contain LaTeX.
_LATEX_TRANSCRIBED = frozenset({"formula_latex"})