File size: 19,906 Bytes
b68816f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
"""Verbatim span validation β€” the primary anti-hallucination control.

Three rules that must not be relaxed:

1. **Normalise whitespace only.** Not case, not punctuation, not diacritics.
   Every additional normalisation is a hole a fabrication can fit through.
2. **On failure the FIELD becomes `None`** and the rejection is recorded, so a
   reviewer can see what the control caught rather than only what it let past.
3. **Never repair a failed span** by rewriting it to something that does match.
   A repaired span is an unfalsifiable claim, which is precisely what this
   control exists to prevent.

If the provenance span itself is not verbatim, *every* guarded field on the
entry is rejected: the entry's only link to evidence is broken, so nothing on it
can be trusted.
"""

from __future__ import annotations

import re
import unicodedata

from ..models import Branch, Chunk, RejectedField

# Fields carrying a factual claim, and therefore span-guarded.
GUARDED_FIELDS: dict[str, tuple[str, ...]] = {
    # v2 dropped `formula_latex` and `interpretation` from the glossary entry β€”
    # the first is now a reference to the formula branch, the second moved to
    # the Interpretation Pack. Neither is unguarded now; both are simply no
    # longer claimed here.
    "glossary": ("definition", "full_name", "source_wording"),
    "rule": ("statement", "condition", "consequence"),
    "formula": ("formula_latex",),
    # Was `()` until 2026-09-02: the branch asked the model to SUMMARISE, and a
    # summary is unverifiable by construction. It now asks the model to LOCATE,
    # and a located quote is checkable like any other. Both fields are
    # transcriptions, so both are also in `_TRANSCRIBED` below.
    "summary": ("title", "purpose_verbatim"),
}


def normalise_ws(text: str) -> str:
    return re.sub(r"\s+", " ", unicodedata.normalize("NFKC", text)).strip()


def span_present(span: str, source: str) -> bool:
    if not span or not span.strip():
        return False
    return normalise_ws(span) in normalise_ws(source)


# Commands that change how a token is SET, never what it means. Unwrapped to
# their contents, so `\mathrm{Total}` and `Total` compare equal.
#
# `mathrm`, `mathsf` and `text` are the three this corpus actually emits, across
# 44 real strings (11 equations x 4 backends). The rest are kept because they
# are the same class of command and cost nothing β€” but they are unobserved here,
# so do not treat their presence as evidence of anything.
_PRESENTATIONAL = (
    "mathrm", "mathsf", "text", "mathbf", "mathit", "textrm", "operatorname"
)

# Standalone style commands: no argument, no meaning. `\displaystyle` is the one
# this corpus emits (pipeline, equation 4).
#
# The boundary is `(?![A-Za-z])`, NOT `\b`. `\b` does not match between `s` and
# `_`, because `_` is a word character β€” so `\limits_{i}` kept its `\limits`
# while `\limits x` lost it. Exactly the bug X22 found in the eval scorer, in a
# different file. Anywhere a LaTeX command can be followed by a subscript, `\b`
# is the wrong boundary.
_BARE_STYLE = re.compile(r"\\displaystyle(?![A-Za-z])|\\limits(?![A-Za-z])|\\!")

# Spacing, in every form LaTeX spells it. All equivalent to nothing here.
# The negative lookbehind matters: without it the `\ ` rule eats one half of a
# `\\` row separator that happens to be followed by a space, conflating a
# structural marker with a cosmetic one.
_SPACING = re.compile(r"~|\\[,;:]|\\quad\b|\\qquad\b|(?<!\\)\\(?=\s)")

# `\begin{array}{lcr}` β€” the alignment spec is presentational; the same array
# arrives as `{l}` from one backend and `{r}` from another.
_ARRAY_ALIGN = re.compile(r"(\\begin\{array\})\{[lcr|@\s]*\}")

# A row separator immediately before `\end{array}` is a trailing empty row.
# Separators BETWEEN rows are structural and are left alone.
_TRAILING_ROW = re.compile(r"(?:\\\\)+(?=\\end\{array\})")

# `\left(` / `
# `\left(` / `\right)` size a delimiter; they do not change what it groups.
_SIZING = re.compile(r"\\left(?=[([{|.])|\\right(?=[)\\]}|.])")

# Every spelling of a fraction that takes two braced arguments. `\dfrac` and
# `\tfrac` differ from `\frac` only in display size. Plain-TeX `{a \over b}` is
# NOT handled: it is infix, so it needs a different parse, and MinerU has not
# been observed emitting it. Unhandled means a false REJECTION, never a false
# acceptance β€” the fail-safe direction.
_FRACTIONS = ("\\frac{", "\\dfrac{", "\\tfrac{")

# ⚠️ CORPUS-DERIVED ASSUMPTION, not a fact about LaTeX.
#
# In LaTeX generally, `\mathrm{x}` is a roman-set VARIABLE named x. Here it is
# read as a multiplication sign, because that is how MinerU transcribes the `Γ—`
# glyph in this corpus: the real PA formula ends `\mathrm{x}100\%` under one
# backend and `\mathsf{x}100\%` under another.
#
# Why it is tolerable: the rewrite is applied to BOTH sides of the comparison,
# so a document that really does have a variable `x` set in roman converts it
# identically in the claim and in the source, and they still match. The residual
# risk is a contrived false acceptance (`Ma\times Capacity` matching a source
# reading `MaxCapacity`), not a false rejection.
#
# Revisit this first if the pipeline ever meets a corpus where `x` is a genuine
# variable name β€” algebra or statistics rather than mining parameters.
_WRAPPED_TIMES = re.compile(
    r"\\(?:" + "|".join(_PRESENTATIONAL) + r")\{[xX]\}"
)


def _unwrap_command(text: str, name: str) -> str:
    """Replace every `\\name{...}` with its contents, braces consumed.

    A balanced scanner rather than `\\{([^{}]*)\\}`, for the reason recorded in
    calibration Β§12: a regex character class cannot match a nested argument and
    fails SILENTLY, which is how `\\frac` lost its division there. The same
    mistake is available here and would have the same shape.
    """
    token = "\\" + name
    out, i = [], 0
    while True:
        start = text.find(token + "{", i)
        if start == -1:
            out.append(text[i:])
            return "".join(out)
        out.append(text[i:start])
        depth, j = 0, start + len(token)
        for j in range(start + len(token), len(text)):
            if text[j] == "{":
                depth += 1
            elif text[j] == "}":
                depth -= 1
                if depth == 0:
                    break
        else:
            # Unbalanced: leave the remainder untouched rather than guess.
            out.append(text[start:])
            return "".join(out)
        out.append(_unwrap_command(text[start + len(token) + 1 : j], name))
        i = j + 1


def _collapse_redundant_parens(text: str) -> str:
    """`((x))` -> `(x)`, where the outer pair wraps nothing but the inner pair.

    Redundant grouping only. A pair that encloses anything else keeps both,
    because `(a+b)/c` and `a+(b/c)` are different formulas and the parentheses
    are the only thing saying so.
    """
    while True:
        out, changed, i = [], False, 0
        while i < len(text):
            if text[i] != "(":
                out.append(text[i])
                i += 1
                continue
            depth, j = 0, i
            for j in range(i, len(text)):
                if text[j] == "(":
                    depth += 1
                elif text[j] == ")":
                    depth -= 1
                    if depth == 0:
                        break
            else:
                out.append(text[i:])
                i = len(text)
                break
            inner = text[i + 1 : j]
            if inner.startswith("(") and inner.endswith(")"):
                d, ok = 0, True
                for k, ch in enumerate(inner):
                    d += (ch == "(") - (ch == ")")
                    if d == 0 and k != len(inner) - 1:
                        ok = False
                        break
                if ok:  # the outer pair wraps exactly one group
                    out.append(inner)
                    i = j + 1
                    changed = True
                    continue
            out.append("(" + inner + ")")
            i = j + 1
        text = "".join(out)
        if not changed:
            return text


def _rewrite_fractions(text: str) -> str:
    """`\\frac{A}{B}` -> `(A)/(B)`, recursively, balanced.

    Canonicalised to the *prose* spelling of division on purpose. The model
    reads rendered prose β€” the document writes `Qty = (INPR Hours) / (Total
    Hours)` β€” so it answers with a slash while the markup carries `\\frac`.
    Both mean the same thing, and rejecting one for not being the other is the
    identity trap this whole function exists to avoid.

    The explicit parentheses are what keep it honest. Dropping them would make
    `\\frac{a+b}{c}` and `a+\\frac{b}{c}` normalise alike, which is a guard
    that cannot tell two different formulas apart. And X21 is still caught:
    its corruption leaves NO division operator at all, not a differently
    spelled one.
    """
    hits = [(text.find(tok), tok) for tok in _FRACTIONS]
    hits = [(at, tok) for at, tok in hits if at != -1]
    if not hits:
        return text
    idx, token = min(hits)
    args, i = [], idx + len(token) - 1  # -1: the token carries its `{`
    skip = idx + len(token) - 1
    for _ in range(2):
        if i >= len(text) or text[i] != "{":
            return text[:skip] + _rewrite_fractions(text[skip:])
        depth = 0
        for j in range(i, len(text)):
            if text[j] == "{":
                depth += 1
            elif text[j] == "}":
                depth -= 1
                if depth == 0:
                    break
        else:
            return text
        args.append(text[i + 1 : j])
        i = j + 1
    numerator, denominator = (_rewrite_fractions(a) for a in args)
    body = f"({numerator})/({denominator})"
    return text[:idx] + body + _rewrite_fractions(text[i:])


def normalise_latex(text: str) -> str:
    """Canonical form for comparing a LaTeX claim against source markup.

    A different notion of equality from `normalise_ws`, not a weaker one. The
    prompt asks the model to *transcribe into* LaTeX β€” a translation β€” so
    testing string identity rejects correct answers written in a different but
    equivalent notation. Every rule below erases a difference that carries no
    meaning in LaTeX, and none erases structure.

    Never applied to prose. In prose whitespace is a word boundary and deleting
    it WOULD open a hole: "no data" would match "nodata".

    Each rule is grounded in markup this corpus actually emits (calibration Β§12
    records the strings): MinerU spaces every character, wraps tokens in
    `\\mathrm{}`, writes `~` for a space, and β€” in the `pipeline` backend β€”
    writes multiplication as `\\mathrm{x}` or as a bare letter `x`.

    **Two known over-normalisations**, both measured rather than suspected, both
    accepted for this corpus and neither safe to assume elsewhere:

    * **Whitespace inside `\\text{}` is deleted along with the rest.** In text
      mode a space is a real word boundary, so `\\text{Total Hours}` and
      `\\text{TotalHours}` compare equal here. Two labels differing only by
      internal spacing therefore collide. Tolerated because a fabrication that
      differs from the truth by nothing but a space is not a threat worth the
      complexity of a mode-aware scanner β€” but it IS a loss of discrimination,
      not a neutral simplification.
    * **`\\mathrm{x}` is read as multiplication** β€” see `_WRAPPED_TIMES` for why
      that is a fact about this corpus rather than about LaTeX.

    What is deliberately NOT normalised, so the guard keeps its teeth:
    structure (`\\frac` is division and its absence is a different formula),
    precedence (the parentheses `\\frac` expands into), operand order, sub- vs
    superscript, `+` vs `-`, and `\\cdot` vs `\\times`.
    """
    out = unicodedata.normalize("NFKC", text)
    # Explicit spacing first, while the whitespace it is defined against still
    # exists: `\ ` is a backslash followed by a space, and is unrecognisable
    # once the spaces are gone.
    out = _SPACING.sub("", out)
    # Then all remaining whitespace, BEFORE any structural rewrite. MinerU
    # writes `\mathrm { P A }` and `\frac { a } { b }` β€” spaced between the
    # command and its brace β€” so a scanner looking for `\mathrm{` finds
    # nothing until this has run. Getting this order wrong does not raise; it
    # silently skips every rewrite and compares raw markup.
    out = re.sub(r"\s+", "", out)
    # A wrapped roman `x` is this document's multiplication sign, not a
    # variable: the real PA formula ends `\mathrm{x}100\%` under one backend and
    # `\mathsf{x}100\%` under another. Must precede the unwrap, which would
    # otherwise destroy the distinction.
    out = _WRAPPED_TIMES.sub(r"\\times", out)
    out = _BARE_STYLE.sub("", out)
    out = _ARRAY_ALIGN.sub(r"\1", out)
    out = _TRAILING_ROW.sub("", out)
    for name in _PRESENTATIONAL:
        out = _unwrap_command(out, name)
    # `\cdot` is deliberately NOT folded into `\times`. It was, speculatively β€”
    # it appears nowhere in the 44 real strings β€” and it is wrong in general: on
    # vectors they are different operations, dot product versus cross product.
    # A guard that cannot tell those apart is worse than one that rejects a
    # `\cdot` it has never actually seen.
    out = out.replace("\\%", "%").replace("$", "")
    out = _SIZING.sub("", out)
    out = _rewrite_fractions(out)
    out = out.replace("{", "").replace("}", "")
    return _collapse_redundant_parens(out)


# Every way this corpus writes multiplication once normalised.
_MULTIPLICATION = re.compile(r"\\times")


def multiplication_is_recoverable(source: str) -> bool:
    """Whether the source distinguishes a multiplication sign at all.

    The `pipeline` backend transcribes `Γ—` as a bare letter `x`, with no
    `\\times` and no `\\mathrm` to mark it β€” calibration Β§12 records this as an
    open question, because `x` is also a plausible variable name. So on such an
    artifact the operator is genuinely absent from the source, and a claim that
    spells it out cannot be checked either way.

    Reported rather than guessed: resolving `x` to multiplication here would
    silently settle a decision that is deliberately still open, and would let
    `Ma\\times Capacity` match a source reading `MaxCapacity`.
    """
    return bool(_MULTIPLICATION.search(normalise_latex(source)))


def latex_present(span: str, source: str) -> bool:
    if not span or not span.strip():
        return False
    return normalise_latex(span) in normalise_latex(source)


def latex_source(chunks: list[Chunk]) -> str:
    """The markup a `formula_latex` claim is validated against."""
    return "\n".join(fragment for chunk in chunks for fragment in chunk.latex)


def evidence_text(chunk_ids: list[str], chunks: list[Chunk]) -> str:
    """The text a span is validated against.

    Includes each chunk's HEADING as well as its body. The heading is part of
    the source document and is often where a term is formally named β€” the
    reference standard heads a section "Physical of Availability (PA)" while
    the body never repeats the phrase. Excluding it would reject a correct,
    verbatim quotation of the document's own section title, which is precisely
    the wording we are required to preserve.
    """
    by_id = {c.chunk_id: c for c in chunks}
    parts: list[str] = []
    for chunk_id in chunk_ids:
        chunk = by_id.get(chunk_id)
        if chunk is None:
            continue
        if chunk.heading:
            parts.append(chunk.heading)
        parts.append(chunk.text)
    return "\n".join(parts)


def _mark_latex(entry, verdict: str) -> None:
    """Record which guarantee the stored `formula_latex` carries.

    A never-throw seam like the rest of validation: an entry type without the
    field is left alone rather than raising, so a branch that gains a LaTeX
    field later opts in by declaring it.
    """
    if hasattr(entry, "latex_verification"):
        entry.latex_verification = verdict


def validate_entry(
    entry, branch: Branch, source_text: str, label: str, latex_text: str = ""
) -> tuple[object, list[RejectedField]]:
    """Returns `(entry, rejections)`; the entry is mutated in place.

    Also checks each guarded field's own value against the source where the
    field is expected to be quoted: `source_wording` and `full_name` are
    literal transcriptions, so a value that cannot be located is a silent
    normalisation β€” exactly the failure this pipeline is required to surface.
    """
    rejections: list[RejectedField] = []
    guarded = GUARDED_FIELDS.get(branch, ())
    span_ok = span_present(entry.provenance.span, source_text)

    for field in guarded:
        value = getattr(entry, field, None)
        if value is None:
            continue

        if not span_ok:
            reason = "provenance.span not found verbatim in evidence"
        elif field in _TRANSCRIBED and not span_present(str(value), source_text):
            reason = f"{field} is not a verbatim transcription of the source"
        elif field in _LATEX_TRANSCRIBED and not latex_text.strip():
            # No markup to compare against. Guarded by provenance alone, and
            # said so on the entry rather than left to look like the strong
            # guarantee.
            _mark_latex(entry, "unverified_no_markup")
            continue
        elif field in _LATEX_TRANSCRIBED:
            # Checked against the SOURCE MARKUP, not `text`. Only reachable when
            # the artifact actually carried markup for these chunks: an artifact
            # written before `Chunk.latex` crossed the seam has nothing to prove
            # the claim against, and "cannot verify" must not present as "failed
            # verification" β€” that would null every formula entry on every older
            # artifact. In that case the field stays guarded by provenance alone,
            # exactly as it was before this check existed.
            if latex_present(str(value), latex_text):
                _mark_latex(entry, "verified")
                continue
            # Same distinction one level down. A claim that spells out a
            # multiplication sign the source never distinguishes is unprovable
            # rather than wrong β€” `pipeline` writes `MOHH x Qty x PA` with no
            # operator to compare against. Rejecting here would null every
            # product formula on every `pipeline` artifact, silently, and it
            # would read as a bad model.
            if _MULTIPLICATION.search(normalise_latex(str(value))) and (
                not multiplication_is_recoverable(latex_text)
            ):
                _mark_latex(entry, "unverified_operator")
                continue
            reason = f"{field} is not a transcription of the source markup"
        else:
            continue

        rejections.append(
            RejectedField(
                entry_term=label,
                field=field,
                offending_value=str(value)[:300],
                reason=reason,
                branch=branch,
            )
        )
        setattr(entry, field, None)

    return entry, rejections


# Fields that claim to be copied from the document word for word. A definition
# may legitimately be assembled across sentences; a "full name" may not.
_TRANSCRIBED = frozenset({"full_name", "source_wording", "title", "purpose_verbatim"})

# Fields that claim to transcribe the document's MARKUP. Checked against
# `Chunk.latex` with `latex_present`, never against `text` β€” putting one of
# these in `_TRANSCRIBED` instead would reject every formula entry, because
# `text` holds rendered prose and can never contain LaTeX.
_LATEX_TRANSCRIBED = frozenset({"formula_latex"})