Download src/knowledge_extraction/cluster/normalize.py from DataEyond/Agentic-Service-Data-Eyond-Catalog: direct link, hf CLI and curl.
- Browser
- Download file 10.8 kB
-
https://huggingface.co/spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/resolve/main/src/knowledge_extraction/cluster/normalize.py
- Command line
-
hf download hf://spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/src/knowledge_extraction/cluster/normalize.py
-
curl -L -o normalize.py https://huggingface.co/spaces/DataEyond/Agentic-Service-Data-Eyond-Catalog/resolve/main/src/knowledge_extraction/cluster/normalize.py
10.8 kB
| """Surface normalisation and the abbreviation index used by clustering. | |
| **This normalisation is for clustering only.** Span validation normalises | |
| whitespace and nothing else — every additional normalisation there is a hole a | |
| fabrication can fit through. Do not reuse `normalize()` in that path. | |
| """ | |
| from __future__ import annotations | |
| import re | |
| import unicodedata | |
| from ..models import AbbrevPair | |
| # Surfaces that carry no discriminating power on their own. A mention of just | |
| # "unit" or "parameter" is not a term. These are dropped as WHOLE surface forms | |
| # only, never as substrings — so no term containing them is ever lost. | |
| STOP_SURFACES = { | |
| "unit", | |
| "type", | |
| "class", | |
| "equipment", | |
| "equipment unit", | |
| "parameter", | |
| "activity", | |
| "data", | |
| "nilai", | |
| "proses", | |
| "hasil", | |
| "total", | |
| } | |
| # Plural folding for the CLUSTERING KEY ONLY (see `normalize(lemma=...)` below). | |
| # Measured 2026-09-10: of the 140 merges the retired fuzzy pass made across three | |
| # documents, 131 were token-subset matches (all wrong) and of the 9 genuine | |
| # near-misses SIX were plain singular/plural pairs — `resource`/`resources`, | |
| # `ore type`/`ore types`, `deposit`/`deposits`, `measurement`/`measurements`, | |
| # `project manager`/`project managers`, `reserve`/`reserves`. Folding those | |
| # deterministically is what let the fuzzy pass be retired instead of guarded: on | |
| # the Open Pit textbook it gives 431 clusters against 434 for the best guarded | |
| # fuzzy config and 445 for no-fuzzy-without-lemma. | |
| # | |
| # Deliberately hand-rolled rather than a real lemmatiser: a proper one means a | |
| # new dependency (nltk/spacy) for a measured six-pair gain, and this runs on a | |
| # normalised key where the only job is collapsing an English plural suffix. | |
| _PLURAL_EXCEPTIONS = ("ss", "us", "is", "as", "os") | |
| def depluralise(token: str) -> str: | |
| """Collapse an English plural suffix on a single normalised token. | |
| Conservative by construction — it only ever SHORTENS a token, and only on | |
| suffixes that are plural in this corpus. `status`, `analysis`, `gas` and | |
| `bias` are protected by `_PLURAL_EXCEPTIONS`; anything under 4 characters is | |
| left alone so an abbreviation (`UAS`, `BCM`) is never touched. | |
| """ | |
| for suffix, replacement in (("ies", "y"), ("ses", "s"), ("xes", "x"), ("hes", "h")): | |
| if token.endswith(suffix) and len(token) > len(suffix) + 1: | |
| return token[: -len(suffix)] + replacement | |
| if ( | |
| token.endswith("s") | |
| and not token.endswith(_PLURAL_EXCEPTIONS) | |
| and len(token) > 3 | |
| ): | |
| return token[:-1] | |
| return token | |
| def normalize(surface: str, lemma: bool = False) -> str: | |
| """Normalise a surface for comparison. | |
| `lemma=True` additionally folds English plurals, and is passed ONLY by the | |
| clustering key path (`is_noise` and `AbbrevIndex`). It is deliberately NOT | |
| the default: `rank/evidence.py` imports this function directly to match a | |
| term against a chunk heading, and the 2026-09-10 measurement that justified | |
| plural folding patched this module's global — which that direct import | |
| bypasses. Making it the default would therefore change evidence ranking in a | |
| way nothing has measured. Keep the ranker on the unfolded form until there | |
| is a number for it. | |
| """ | |
| s = unicodedata.normalize("NFKC", surface).casefold() | |
| s = s.replace("-", " ").replace("_", " ") | |
| s = re.sub(r"[.’']", "", s) | |
| s = re.sub(r"[^\w\s/()]", " ", s) | |
| s = re.sub(r"\s+", " ", s) | |
| s = s.strip(" ()/") | |
| # Drop an UNBALANCED bracket rather than leave it in the key. The edge strip | |
| # above is what CREATES the imbalance: "Utilization of Availability (UA)" | |
| # lost its trailing ")" and kept the opening "(", leaving | |
| # "utilization of availability (ua" — a malformed key that matched nothing | |
| # and gave one concept two clusters. Checked after the strip, not before. | |
| if s.count("(") != s.count(")"): | |
| s = re.sub(r"\s+", " ", s.replace("(", " ").replace(")", " ")).strip() | |
| # Applied LAST, per token, so it never interferes with the bracket and | |
| # punctuation handling above. | |
| if lemma: | |
| s = " ".join(depluralise(t) for t in s.split()) | |
| return s | |
| def is_noise(surface: str) -> bool: | |
| n = normalize(surface, lemma=True) | |
| if len(n) < 2: | |
| return True | |
| if n in STOP_SURFACES: | |
| return True | |
| return not re.search(r"[a-z]", n) # pure numbers / symbols | |
| _CONNECTORS = {"of", "the", "and", "for", "in", "on", "to", "a", "an", | |
| "dan", "untuk", "pada", "di", "ke", "yang", "dari"} | |
| # "Utilization of Availability (UA)" -> ("Utilization of Availability", "UA"). | |
| # The inner group is bounded: a long parenthetical is a clarification, not an | |
| # abbreviation, and must not be treated as one. | |
| _PARENTHETICAL = re.compile(r"^(?P<outer>.+?)\s*\((?P<inner>[^()]{1,16})\)$") | |
| def split_parenthetical(surface: str) -> tuple[str, str] | None: | |
| """Split a trailing bracketed form off a surface, if there is one.""" | |
| match = _PARENTHETICAL.match(surface.strip()) | |
| if not match: | |
| return None | |
| outer, inner = match.group("outer").strip(), match.group("inner").strip() | |
| return (outer, inner) if outer and inner else None | |
| def looks_like_abbreviation(text: str) -> bool: | |
| """Conservative: short, no spaces, and not sentence-case prose. | |
| The guard on the fallback below. "Production (planned)" must NOT be read as | |
| an abbreviation pair, or every qualified form silently merges into its base | |
| term and a real distinction is lost. | |
| """ | |
| t = text.strip() | |
| if not t or " " in t or len(t) > 8: | |
| return False | |
| letters = [c for c in t if c.isalpha()] | |
| return bool(letters) and sum(c.isupper() for c in letters) >= max( | |
| 1, len(letters) - 1 | |
| ) | |
| def is_initialism(abbrev: str, expansion: str) -> bool: | |
| """Does `abbrev` spell the initials of `expansion`? | |
| Lets a term that declares its own abbreviation inline — the common shape in | |
| technical standards — cluster correctly even when the legend block never | |
| listed the pair. | |
| """ | |
| a = "".join(c for c in abbrev if c.isalpha()).casefold() | |
| if len(a) < 2: | |
| return False | |
| words = re.findall(r"[^\W\d_]+", expansion, flags=re.UNICODE) | |
| if not words: | |
| return False | |
| every = "".join(w[0] for w in words).casefold() | |
| significant = "".join( | |
| w[0] for w in words if w.casefold() not in _CONNECTORS | |
| ).casefold() | |
| return a in (every, significant) | |
| class AbbrevIndex: | |
| """Bidirectional abbreviation ↔ expansion lookup built from legend blocks. | |
| This is why the legend filter runs before clustering: without it, `PA` and | |
| `Physical Availability` never meet. | |
| """ | |
| def __init__(self, pairs: list[AbbrevPair]): | |
| self.to_expansion: dict[str, str] = {} | |
| self.to_abbrev: dict[str, str] = {} | |
| for pair in pairs: | |
| abbrev = normalize(pair.abbrev, lemma=True) | |
| expansion = normalize(pair.expansion, lemma=True) | |
| if not abbrev or not expansion: | |
| continue | |
| self.to_expansion[abbrev] = expansion | |
| self.to_abbrev[expansion] = abbrev | |
| # Words that carry no initial in an initialism: "Utilization **of** | |
| # Availability" is UA, not UOA. Indonesian connectors included because the | |
| # corpus is bilingual. | |
| def learn_inline(self, surfaces) -> int: | |
| """Register `X (Y)` pairs a document declares inline. Returns the count. | |
| A standard that writes "Waste Removal (WR)" has declared the pair as | |
| surely as a legend block would, and the corpus does this constantly. It | |
| matters for a reason that is easy to miss: keying only the BRACKETED | |
| form to the abbreviation splits the concept three ways instead of one, | |
| because bare "Waste Removal" and bare "WR" still key to themselves and | |
| nothing links them. Registering the pair links all three. | |
| Legend entries win — they are the document's own authority — so an | |
| inline pair never overwrites one. | |
| """ | |
| learned = 0 | |
| for surface in surfaces: | |
| parts = split_parenthetical(surface) | |
| if not parts: | |
| continue | |
| outer, inner = parts | |
| if looks_like_abbreviation(inner) and is_initialism(inner, outer): | |
| abbrev, expansion = normalize(inner, lemma=True), normalize(outer, lemma=True) | |
| elif looks_like_abbreviation(outer) and is_initialism(outer, inner): | |
| abbrev, expansion = normalize(outer, lemma=True), normalize(inner, lemma=True) | |
| else: | |
| continue | |
| if abbrev and expansion and abbrev not in self.to_expansion: | |
| self.to_expansion[abbrev] = expansion | |
| self.to_abbrev[expansion] = abbrev | |
| learned += 1 | |
| return learned | |
| def canonical_key(self, surface: str) -> str: | |
| """Map a surface to a shared key so an abbreviation and its expansion | |
| collide into the same bucket.""" | |
| n = normalize(surface, lemma=True) | |
| if n in self.to_abbrev: | |
| return self.to_abbrev[n] | |
| if n in self.to_expansion: | |
| return n | |
| # "X (Y)" — the shape that produced two clusters for one concept. UA and | |
| # "Utilization of Availability (UA)" were separate terms, each with its | |
| # own formula, because the bracketed form matched neither the legend's | |
| # expansion nor the bare abbreviation. | |
| parts = split_parenthetical(surface) | |
| if parts: | |
| outer, inner = parts | |
| outer_n, inner_n = normalize(outer, lemma=True), normalize(inner, lemma=True) | |
| if inner_n in self.to_expansion: # the legend knows the bracket | |
| return inner_n | |
| if outer_n in self.to_abbrev: # the legend knows the outer | |
| return self.to_abbrev[outer_n] | |
| if looks_like_abbreviation(inner) and is_initialism(inner, outer): | |
| return inner_n # self-declaring, no legend | |
| if looks_like_abbreviation(outer) and is_initialism(outer, inner): | |
| return outer_n # "UA (Utilization of ...)" | |
| if looks_like_abbreviation(inner): | |
| return outer_n # bracket is an alias; key on the term | |
| return n | |
| def linked(self, a: str, b: str) -> bool: | |
| na, nb = normalize(a, lemma=True), normalize(b, lemma=True) | |
| if self.to_expansion.get(na) == nb or self.to_expansion.get(nb) == na: | |
| return True | |
| # Same reasoning as `canonical_key`: two surfaces are linked when one is | |
| # the other's bracketed form. | |
| return self.canonical_key(a) == self.canonical_key(b) | |