ishaq101's picture
/fix parsing and term extract (#21)
f07443e
Raw History Blame Contribute Delete
10.8 kB
"""Surface normalisation and the abbreviation index used by clustering.
**This normalisation is for clustering only.** Span validation normalises
whitespace and nothing else — every additional normalisation there is a hole a
fabrication can fit through. Do not reuse `normalize()` in that path.
"""
from __future__ import annotations
import re
import unicodedata
from ..models import AbbrevPair
# Surfaces that carry no discriminating power on their own. A mention of just
# "unit" or "parameter" is not a term. These are dropped as WHOLE surface forms
# only, never as substrings — so no term containing them is ever lost.
STOP_SURFACES = {
"unit",
"type",
"class",
"equipment",
"equipment unit",
"parameter",
"activity",
"data",
"nilai",
"proses",
"hasil",
"total",
}
# Plural folding for the CLUSTERING KEY ONLY (see `normalize(lemma=...)` below).
# Measured 2026-09-10: of the 140 merges the retired fuzzy pass made across three
# documents, 131 were token-subset matches (all wrong) and of the 9 genuine
# near-misses SIX were plain singular/plural pairs — `resource`/`resources`,
# `ore type`/`ore types`, `deposit`/`deposits`, `measurement`/`measurements`,
# `project manager`/`project managers`, `reserve`/`reserves`. Folding those
# deterministically is what let the fuzzy pass be retired instead of guarded: on
# the Open Pit textbook it gives 431 clusters against 434 for the best guarded
# fuzzy config and 445 for no-fuzzy-without-lemma.
#
# Deliberately hand-rolled rather than a real lemmatiser: a proper one means a
# new dependency (nltk/spacy) for a measured six-pair gain, and this runs on a
# normalised key where the only job is collapsing an English plural suffix.
_PLURAL_EXCEPTIONS = ("ss", "us", "is", "as", "os")
def depluralise(token: str) -> str:
"""Collapse an English plural suffix on a single normalised token.
Conservative by construction — it only ever SHORTENS a token, and only on
suffixes that are plural in this corpus. `status`, `analysis`, `gas` and
`bias` are protected by `_PLURAL_EXCEPTIONS`; anything under 4 characters is
left alone so an abbreviation (`UAS`, `BCM`) is never touched.
"""
for suffix, replacement in (("ies", "y"), ("ses", "s"), ("xes", "x"), ("hes", "h")):
if token.endswith(suffix) and len(token) > len(suffix) + 1:
return token[: -len(suffix)] + replacement
if (
token.endswith("s")
and not token.endswith(_PLURAL_EXCEPTIONS)
and len(token) > 3
):
return token[:-1]
return token
def normalize(surface: str, lemma: bool = False) -> str:
"""Normalise a surface for comparison.
`lemma=True` additionally folds English plurals, and is passed ONLY by the
clustering key path (`is_noise` and `AbbrevIndex`). It is deliberately NOT
the default: `rank/evidence.py` imports this function directly to match a
term against a chunk heading, and the 2026-09-10 measurement that justified
plural folding patched this module's global — which that direct import
bypasses. Making it the default would therefore change evidence ranking in a
way nothing has measured. Keep the ranker on the unfolded form until there
is a number for it.
"""
s = unicodedata.normalize("NFKC", surface).casefold()
s = s.replace("-", " ").replace("_", " ")
s = re.sub(r"[.’']", "", s)
s = re.sub(r"[^\w\s/()]", " ", s)
s = re.sub(r"\s+", " ", s)
s = s.strip(" ()/")
# Drop an UNBALANCED bracket rather than leave it in the key. The edge strip
# above is what CREATES the imbalance: "Utilization of Availability (UA)"
# lost its trailing ")" and kept the opening "(", leaving
# "utilization of availability (ua" — a malformed key that matched nothing
# and gave one concept two clusters. Checked after the strip, not before.
if s.count("(") != s.count(")"):
s = re.sub(r"\s+", " ", s.replace("(", " ").replace(")", " ")).strip()
# Applied LAST, per token, so it never interferes with the bracket and
# punctuation handling above.
if lemma:
s = " ".join(depluralise(t) for t in s.split())
return s
def is_noise(surface: str) -> bool:
n = normalize(surface, lemma=True)
if len(n) < 2:
return True
if n in STOP_SURFACES:
return True
return not re.search(r"[a-z]", n) # pure numbers / symbols
_CONNECTORS = {"of", "the", "and", "for", "in", "on", "to", "a", "an",
"dan", "untuk", "pada", "di", "ke", "yang", "dari"}
# "Utilization of Availability (UA)" -> ("Utilization of Availability", "UA").
# The inner group is bounded: a long parenthetical is a clarification, not an
# abbreviation, and must not be treated as one.
_PARENTHETICAL = re.compile(r"^(?P<outer>.+?)\s*\((?P<inner>[^()]{1,16})\)$")
def split_parenthetical(surface: str) -> tuple[str, str] | None:
"""Split a trailing bracketed form off a surface, if there is one."""
match = _PARENTHETICAL.match(surface.strip())
if not match:
return None
outer, inner = match.group("outer").strip(), match.group("inner").strip()
return (outer, inner) if outer and inner else None
def looks_like_abbreviation(text: str) -> bool:
"""Conservative: short, no spaces, and not sentence-case prose.
The guard on the fallback below. "Production (planned)" must NOT be read as
an abbreviation pair, or every qualified form silently merges into its base
term and a real distinction is lost.
"""
t = text.strip()
if not t or " " in t or len(t) > 8:
return False
letters = [c for c in t if c.isalpha()]
return bool(letters) and sum(c.isupper() for c in letters) >= max(
1, len(letters) - 1
)
def is_initialism(abbrev: str, expansion: str) -> bool:
"""Does `abbrev` spell the initials of `expansion`?
Lets a term that declares its own abbreviation inline — the common shape in
technical standards — cluster correctly even when the legend block never
listed the pair.
"""
a = "".join(c for c in abbrev if c.isalpha()).casefold()
if len(a) < 2:
return False
words = re.findall(r"[^\W\d_]+", expansion, flags=re.UNICODE)
if not words:
return False
every = "".join(w[0] for w in words).casefold()
significant = "".join(
w[0] for w in words if w.casefold() not in _CONNECTORS
).casefold()
return a in (every, significant)
class AbbrevIndex:
"""Bidirectional abbreviation ↔ expansion lookup built from legend blocks.
This is why the legend filter runs before clustering: without it, `PA` and
`Physical Availability` never meet.
"""
def __init__(self, pairs: list[AbbrevPair]):
self.to_expansion: dict[str, str] = {}
self.to_abbrev: dict[str, str] = {}
for pair in pairs:
abbrev = normalize(pair.abbrev, lemma=True)
expansion = normalize(pair.expansion, lemma=True)
if not abbrev or not expansion:
continue
self.to_expansion[abbrev] = expansion
self.to_abbrev[expansion] = abbrev
# Words that carry no initial in an initialism: "Utilization **of**
# Availability" is UA, not UOA. Indonesian connectors included because the
# corpus is bilingual.
def learn_inline(self, surfaces) -> int:
"""Register `X (Y)` pairs a document declares inline. Returns the count.
A standard that writes "Waste Removal (WR)" has declared the pair as
surely as a legend block would, and the corpus does this constantly. It
matters for a reason that is easy to miss: keying only the BRACKETED
form to the abbreviation splits the concept three ways instead of one,
because bare "Waste Removal" and bare "WR" still key to themselves and
nothing links them. Registering the pair links all three.
Legend entries win — they are the document's own authority — so an
inline pair never overwrites one.
"""
learned = 0
for surface in surfaces:
parts = split_parenthetical(surface)
if not parts:
continue
outer, inner = parts
if looks_like_abbreviation(inner) and is_initialism(inner, outer):
abbrev, expansion = normalize(inner, lemma=True), normalize(outer, lemma=True)
elif looks_like_abbreviation(outer) and is_initialism(outer, inner):
abbrev, expansion = normalize(outer, lemma=True), normalize(inner, lemma=True)
else:
continue
if abbrev and expansion and abbrev not in self.to_expansion:
self.to_expansion[abbrev] = expansion
self.to_abbrev[expansion] = abbrev
learned += 1
return learned
def canonical_key(self, surface: str) -> str:
"""Map a surface to a shared key so an abbreviation and its expansion
collide into the same bucket."""
n = normalize(surface, lemma=True)
if n in self.to_abbrev:
return self.to_abbrev[n]
if n in self.to_expansion:
return n
# "X (Y)" — the shape that produced two clusters for one concept. UA and
# "Utilization of Availability (UA)" were separate terms, each with its
# own formula, because the bracketed form matched neither the legend's
# expansion nor the bare abbreviation.
parts = split_parenthetical(surface)
if parts:
outer, inner = parts
outer_n, inner_n = normalize(outer, lemma=True), normalize(inner, lemma=True)
if inner_n in self.to_expansion: # the legend knows the bracket
return inner_n
if outer_n in self.to_abbrev: # the legend knows the outer
return self.to_abbrev[outer_n]
if looks_like_abbreviation(inner) and is_initialism(inner, outer):
return inner_n # self-declaring, no legend
if looks_like_abbreviation(outer) and is_initialism(outer, inner):
return outer_n # "UA (Utilization of ...)"
if looks_like_abbreviation(inner):
return outer_n # bracket is an alias; key on the term
return n
def linked(self, a: str, b: str) -> bool:
na, nb = normalize(a, lemma=True), normalize(b, lemma=True)
if self.to_expansion.get(na) == nb or self.to_expansion.get(nb) == na:
return True
# Same reasoning as `canonical_key`: two surfaces are linked when one is
# the other's bracketed form.
return self.canonical_key(a) == self.canonical_key(b)