"""Surface normalisation and the abbreviation index used by clustering. **This normalisation is for clustering only.** Span validation normalises whitespace and nothing else — every additional normalisation there is a hole a fabrication can fit through. Do not reuse `normalize()` in that path. """ from __future__ import annotations import re import unicodedata from ..models import AbbrevPair # Surfaces that carry no discriminating power on their own. A mention of just # "unit" or "parameter" is not a term. These are dropped as WHOLE surface forms # only, never as substrings — so no term containing them is ever lost. STOP_SURFACES = { "unit", "type", "class", "equipment", "equipment unit", "parameter", "activity", "data", "nilai", "proses", "hasil", "total", } # Plural folding for the CLUSTERING KEY ONLY (see `normalize(lemma=...)` below). # Measured 2026-09-10: of the 140 merges the retired fuzzy pass made across three # documents, 131 were token-subset matches (all wrong) and of the 9 genuine # near-misses SIX were plain singular/plural pairs — `resource`/`resources`, # `ore type`/`ore types`, `deposit`/`deposits`, `measurement`/`measurements`, # `project manager`/`project managers`, `reserve`/`reserves`. Folding those # deterministically is what let the fuzzy pass be retired instead of guarded: on # the Open Pit textbook it gives 431 clusters against 434 for the best guarded # fuzzy config and 445 for no-fuzzy-without-lemma. # # Deliberately hand-rolled rather than a real lemmatiser: a proper one means a # new dependency (nltk/spacy) for a measured six-pair gain, and this runs on a # normalised key where the only job is collapsing an English plural suffix. _PLURAL_EXCEPTIONS = ("ss", "us", "is", "as", "os") def depluralise(token: str) -> str: """Collapse an English plural suffix on a single normalised token. Conservative by construction — it only ever SHORTENS a token, and only on suffixes that are plural in this corpus. `status`, `analysis`, `gas` and `bias` are protected by `_PLURAL_EXCEPTIONS`; anything under 4 characters is left alone so an abbreviation (`UAS`, `BCM`) is never touched. """ for suffix, replacement in (("ies", "y"), ("ses", "s"), ("xes", "x"), ("hes", "h")): if token.endswith(suffix) and len(token) > len(suffix) + 1: return token[: -len(suffix)] + replacement if ( token.endswith("s") and not token.endswith(_PLURAL_EXCEPTIONS) and len(token) > 3 ): return token[:-1] return token def normalize(surface: str, lemma: bool = False) -> str: """Normalise a surface for comparison. `lemma=True` additionally folds English plurals, and is passed ONLY by the clustering key path (`is_noise` and `AbbrevIndex`). It is deliberately NOT the default: `rank/evidence.py` imports this function directly to match a term against a chunk heading, and the 2026-09-10 measurement that justified plural folding patched this module's global — which that direct import bypasses. Making it the default would therefore change evidence ranking in a way nothing has measured. Keep the ranker on the unfolded form until there is a number for it. """ s = unicodedata.normalize("NFKC", surface).casefold() s = s.replace("-", " ").replace("_", " ") s = re.sub(r"[.’']", "", s) s = re.sub(r"[^\w\s/()]", " ", s) s = re.sub(r"\s+", " ", s) s = s.strip(" ()/") # Drop an UNBALANCED bracket rather than leave it in the key. The edge strip # above is what CREATES the imbalance: "Utilization of Availability (UA)" # lost its trailing ")" and kept the opening "(", leaving # "utilization of availability (ua" — a malformed key that matched nothing # and gave one concept two clusters. Checked after the strip, not before. if s.count("(") != s.count(")"): s = re.sub(r"\s+", " ", s.replace("(", " ").replace(")", " ")).strip() # Applied LAST, per token, so it never interferes with the bracket and # punctuation handling above. if lemma: s = " ".join(depluralise(t) for t in s.split()) return s def is_noise(surface: str) -> bool: n = normalize(surface, lemma=True) if len(n) < 2: return True if n in STOP_SURFACES: return True return not re.search(r"[a-z]", n) # pure numbers / symbols _CONNECTORS = {"of", "the", "and", "for", "in", "on", "to", "a", "an", "dan", "untuk", "pada", "di", "ke", "yang", "dari"} # "Utilization of Availability (UA)" -> ("Utilization of Availability", "UA"). # The inner group is bounded: a long parenthetical is a clarification, not an # abbreviation, and must not be treated as one. _PARENTHETICAL = re.compile(r"^(?P.+?)\s*\((?P[^()]{1,16})\)$") def split_parenthetical(surface: str) -> tuple[str, str] | None: """Split a trailing bracketed form off a surface, if there is one.""" match = _PARENTHETICAL.match(surface.strip()) if not match: return None outer, inner = match.group("outer").strip(), match.group("inner").strip() return (outer, inner) if outer and inner else None def looks_like_abbreviation(text: str) -> bool: """Conservative: short, no spaces, and not sentence-case prose. The guard on the fallback below. "Production (planned)" must NOT be read as an abbreviation pair, or every qualified form silently merges into its base term and a real distinction is lost. """ t = text.strip() if not t or " " in t or len(t) > 8: return False letters = [c for c in t if c.isalpha()] return bool(letters) and sum(c.isupper() for c in letters) >= max( 1, len(letters) - 1 ) def is_initialism(abbrev: str, expansion: str) -> bool: """Does `abbrev` spell the initials of `expansion`? Lets a term that declares its own abbreviation inline — the common shape in technical standards — cluster correctly even when the legend block never listed the pair. """ a = "".join(c for c in abbrev if c.isalpha()).casefold() if len(a) < 2: return False words = re.findall(r"[^\W\d_]+", expansion, flags=re.UNICODE) if not words: return False every = "".join(w[0] for w in words).casefold() significant = "".join( w[0] for w in words if w.casefold() not in _CONNECTORS ).casefold() return a in (every, significant) class AbbrevIndex: """Bidirectional abbreviation ↔ expansion lookup built from legend blocks. This is why the legend filter runs before clustering: without it, `PA` and `Physical Availability` never meet. """ def __init__(self, pairs: list[AbbrevPair]): self.to_expansion: dict[str, str] = {} self.to_abbrev: dict[str, str] = {} for pair in pairs: abbrev = normalize(pair.abbrev, lemma=True) expansion = normalize(pair.expansion, lemma=True) if not abbrev or not expansion: continue self.to_expansion[abbrev] = expansion self.to_abbrev[expansion] = abbrev # Words that carry no initial in an initialism: "Utilization **of** # Availability" is UA, not UOA. Indonesian connectors included because the # corpus is bilingual. def learn_inline(self, surfaces) -> int: """Register `X (Y)` pairs a document declares inline. Returns the count. A standard that writes "Waste Removal (WR)" has declared the pair as surely as a legend block would, and the corpus does this constantly. It matters for a reason that is easy to miss: keying only the BRACKETED form to the abbreviation splits the concept three ways instead of one, because bare "Waste Removal" and bare "WR" still key to themselves and nothing links them. Registering the pair links all three. Legend entries win — they are the document's own authority — so an inline pair never overwrites one. """ learned = 0 for surface in surfaces: parts = split_parenthetical(surface) if not parts: continue outer, inner = parts if looks_like_abbreviation(inner) and is_initialism(inner, outer): abbrev, expansion = normalize(inner, lemma=True), normalize(outer, lemma=True) elif looks_like_abbreviation(outer) and is_initialism(outer, inner): abbrev, expansion = normalize(outer, lemma=True), normalize(inner, lemma=True) else: continue if abbrev and expansion and abbrev not in self.to_expansion: self.to_expansion[abbrev] = expansion self.to_abbrev[expansion] = abbrev learned += 1 return learned def canonical_key(self, surface: str) -> str: """Map a surface to a shared key so an abbreviation and its expansion collide into the same bucket.""" n = normalize(surface, lemma=True) if n in self.to_abbrev: return self.to_abbrev[n] if n in self.to_expansion: return n # "X (Y)" — the shape that produced two clusters for one concept. UA and # "Utilization of Availability (UA)" were separate terms, each with its # own formula, because the bracketed form matched neither the legend's # expansion nor the bare abbreviation. parts = split_parenthetical(surface) if parts: outer, inner = parts outer_n, inner_n = normalize(outer, lemma=True), normalize(inner, lemma=True) if inner_n in self.to_expansion: # the legend knows the bracket return inner_n if outer_n in self.to_abbrev: # the legend knows the outer return self.to_abbrev[outer_n] if looks_like_abbreviation(inner) and is_initialism(inner, outer): return inner_n # self-declaring, no legend if looks_like_abbreviation(outer) and is_initialism(outer, inner): return outer_n # "UA (Utilization of ...)" if looks_like_abbreviation(inner): return outer_n # bracket is an alias; key on the term return n def linked(self, a: str, b: str) -> bool: na, nb = normalize(a, lemma=True), normalize(b, lemma=True) if self.to_expansion.get(na) == nb or self.to_expansion.get(nb) == na: return True # Same reasoning as `canonical_key`: two surfaces are linked when one is # the other's bracketed form. return self.canonical_key(a) == self.canonical_key(b)