File size: 10,782 Bytes
b68816f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
f07443e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
b68816f
 
 
 
 
 
 
 
 
 
 
 
 
f07443e
 
 
 
b68816f
 
 
 
f07443e
b68816f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
f07443e
 
b68816f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
f07443e
b68816f
f07443e
b68816f
 
 
 
 
 
 
 
 
 
 
f07443e
b68816f
 
 
 
 
 
 
 
 
 
 
 
f07443e
b68816f
 
 
 
 
 
 
 
 
 
 
 
 
f07443e
b68816f
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
"""Surface normalisation and the abbreviation index used by clustering.

**This normalisation is for clustering only.** Span validation normalises
whitespace and nothing else β€” every additional normalisation there is a hole a
fabrication can fit through. Do not reuse `normalize()` in that path.
"""

from __future__ import annotations

import re
import unicodedata

from ..models import AbbrevPair

# Surfaces that carry no discriminating power on their own. A mention of just
# "unit" or "parameter" is not a term. These are dropped as WHOLE surface forms
# only, never as substrings β€” so no term containing them is ever lost.
STOP_SURFACES = {
    "unit",
    "type",
    "class",
    "equipment",
    "equipment unit",
    "parameter",
    "activity",
    "data",
    "nilai",
    "proses",
    "hasil",
    "total",
}


# Plural folding for the CLUSTERING KEY ONLY (see `normalize(lemma=...)` below).
# Measured 2026-09-10: of the 140 merges the retired fuzzy pass made across three
# documents, 131 were token-subset matches (all wrong) and of the 9 genuine
# near-misses SIX were plain singular/plural pairs β€” `resource`/`resources`,
# `ore type`/`ore types`, `deposit`/`deposits`, `measurement`/`measurements`,
# `project manager`/`project managers`, `reserve`/`reserves`. Folding those
# deterministically is what let the fuzzy pass be retired instead of guarded: on
# the Open Pit textbook it gives 431 clusters against 434 for the best guarded
# fuzzy config and 445 for no-fuzzy-without-lemma.
#
# Deliberately hand-rolled rather than a real lemmatiser: a proper one means a
# new dependency (nltk/spacy) for a measured six-pair gain, and this runs on a
# normalised key where the only job is collapsing an English plural suffix.
_PLURAL_EXCEPTIONS = ("ss", "us", "is", "as", "os")


def depluralise(token: str) -> str:
    """Collapse an English plural suffix on a single normalised token.

    Conservative by construction β€” it only ever SHORTENS a token, and only on
    suffixes that are plural in this corpus. `status`, `analysis`, `gas` and
    `bias` are protected by `_PLURAL_EXCEPTIONS`; anything under 4 characters is
    left alone so an abbreviation (`UAS`, `BCM`) is never touched.
    """
    for suffix, replacement in (("ies", "y"), ("ses", "s"), ("xes", "x"), ("hes", "h")):
        if token.endswith(suffix) and len(token) > len(suffix) + 1:
            return token[: -len(suffix)] + replacement
    if (
        token.endswith("s")
        and not token.endswith(_PLURAL_EXCEPTIONS)
        and len(token) > 3
    ):
        return token[:-1]
    return token


def normalize(surface: str, lemma: bool = False) -> str:
    """Normalise a surface for comparison.

    `lemma=True` additionally folds English plurals, and is passed ONLY by the
    clustering key path (`is_noise` and `AbbrevIndex`). It is deliberately NOT
    the default: `rank/evidence.py` imports this function directly to match a
    term against a chunk heading, and the 2026-09-10 measurement that justified
    plural folding patched this module's global β€” which that direct import
    bypasses. Making it the default would therefore change evidence ranking in a
    way nothing has measured. Keep the ranker on the unfolded form until there
    is a number for it.
    """
    s = unicodedata.normalize("NFKC", surface).casefold()
    s = s.replace("-", " ").replace("_", " ")
    s = re.sub(r"[.’']", "", s)
    s = re.sub(r"[^\w\s/()]", " ", s)
    s = re.sub(r"\s+", " ", s)
    s = s.strip(" ()/")
    # Drop an UNBALANCED bracket rather than leave it in the key. The edge strip
    # above is what CREATES the imbalance: "Utilization of Availability (UA)"
    # lost its trailing ")" and kept the opening "(", leaving
    # "utilization of availability (ua" β€” a malformed key that matched nothing
    # and gave one concept two clusters. Checked after the strip, not before.
    if s.count("(") != s.count(")"):
        s = re.sub(r"\s+", " ", s.replace("(", " ").replace(")", " ")).strip()
    # Applied LAST, per token, so it never interferes with the bracket and
    # punctuation handling above.
    if lemma:
        s = " ".join(depluralise(t) for t in s.split())
    return s


def is_noise(surface: str) -> bool:
    n = normalize(surface, lemma=True)
    if len(n) < 2:
        return True
    if n in STOP_SURFACES:
        return True
    return not re.search(r"[a-z]", n)  # pure numbers / symbols


_CONNECTORS = {"of", "the", "and", "for", "in", "on", "to", "a", "an",
               "dan", "untuk", "pada", "di", "ke", "yang", "dari"}

# "Utilization of Availability (UA)" -> ("Utilization of Availability", "UA").
# The inner group is bounded: a long parenthetical is a clarification, not an
# abbreviation, and must not be treated as one.
_PARENTHETICAL = re.compile(r"^(?P<outer>.+?)\s*\((?P<inner>[^()]{1,16})\)$")


def split_parenthetical(surface: str) -> tuple[str, str] | None:
    """Split a trailing bracketed form off a surface, if there is one."""
    match = _PARENTHETICAL.match(surface.strip())
    if not match:
        return None
    outer, inner = match.group("outer").strip(), match.group("inner").strip()
    return (outer, inner) if outer and inner else None


def looks_like_abbreviation(text: str) -> bool:
    """Conservative: short, no spaces, and not sentence-case prose.

    The guard on the fallback below. "Production (planned)" must NOT be read as
    an abbreviation pair, or every qualified form silently merges into its base
    term and a real distinction is lost.
    """
    t = text.strip()
    if not t or " " in t or len(t) > 8:
        return False
    letters = [c for c in t if c.isalpha()]
    return bool(letters) and sum(c.isupper() for c in letters) >= max(
        1, len(letters) - 1
    )


def is_initialism(abbrev: str, expansion: str) -> bool:
    """Does `abbrev` spell the initials of `expansion`?

    Lets a term that declares its own abbreviation inline β€” the common shape in
    technical standards β€” cluster correctly even when the legend block never
    listed the pair.
    """
    a = "".join(c for c in abbrev if c.isalpha()).casefold()
    if len(a) < 2:
        return False
    words = re.findall(r"[^\W\d_]+", expansion, flags=re.UNICODE)
    if not words:
        return False
    every = "".join(w[0] for w in words).casefold()
    significant = "".join(
        w[0] for w in words if w.casefold() not in _CONNECTORS
    ).casefold()
    return a in (every, significant)


class AbbrevIndex:
    """Bidirectional abbreviation ↔ expansion lookup built from legend blocks.

    This is why the legend filter runs before clustering: without it, `PA` and
    `Physical Availability` never meet.
    """

    def __init__(self, pairs: list[AbbrevPair]):
        self.to_expansion: dict[str, str] = {}
        self.to_abbrev: dict[str, str] = {}
        for pair in pairs:
            abbrev = normalize(pair.abbrev, lemma=True)
            expansion = normalize(pair.expansion, lemma=True)
            if not abbrev or not expansion:
                continue
            self.to_expansion[abbrev] = expansion
            self.to_abbrev[expansion] = abbrev

# Words that carry no initial in an initialism: "Utilization **of**
# Availability" is UA, not UOA. Indonesian connectors included because the
# corpus is bilingual.

    def learn_inline(self, surfaces) -> int:
        """Register `X (Y)` pairs a document declares inline. Returns the count.

        A standard that writes "Waste Removal (WR)" has declared the pair as
        surely as a legend block would, and the corpus does this constantly. It
        matters for a reason that is easy to miss: keying only the BRACKETED
        form to the abbreviation splits the concept three ways instead of one,
        because bare "Waste Removal" and bare "WR" still key to themselves and
        nothing links them. Registering the pair links all three.

        Legend entries win β€” they are the document's own authority β€” so an
        inline pair never overwrites one.
        """
        learned = 0
        for surface in surfaces:
            parts = split_parenthetical(surface)
            if not parts:
                continue
            outer, inner = parts
            if looks_like_abbreviation(inner) and is_initialism(inner, outer):
                abbrev, expansion = normalize(inner, lemma=True), normalize(outer, lemma=True)
            elif looks_like_abbreviation(outer) and is_initialism(outer, inner):
                abbrev, expansion = normalize(outer, lemma=True), normalize(inner, lemma=True)
            else:
                continue
            if abbrev and expansion and abbrev not in self.to_expansion:
                self.to_expansion[abbrev] = expansion
                self.to_abbrev[expansion] = abbrev
                learned += 1
        return learned

    def canonical_key(self, surface: str) -> str:
        """Map a surface to a shared key so an abbreviation and its expansion
        collide into the same bucket."""
        n = normalize(surface, lemma=True)
        if n in self.to_abbrev:
            return self.to_abbrev[n]
        if n in self.to_expansion:
            return n

        # "X (Y)" β€” the shape that produced two clusters for one concept. UA and
        # "Utilization of Availability (UA)" were separate terms, each with its
        # own formula, because the bracketed form matched neither the legend's
        # expansion nor the bare abbreviation.
        parts = split_parenthetical(surface)
        if parts:
            outer, inner = parts
            outer_n, inner_n = normalize(outer, lemma=True), normalize(inner, lemma=True)
            if inner_n in self.to_expansion:      # the legend knows the bracket
                return inner_n
            if outer_n in self.to_abbrev:         # the legend knows the outer
                return self.to_abbrev[outer_n]
            if looks_like_abbreviation(inner) and is_initialism(inner, outer):
                return inner_n                    # self-declaring, no legend
            if looks_like_abbreviation(outer) and is_initialism(outer, inner):
                return outer_n                    # "UA (Utilization of ...)"
            if looks_like_abbreviation(inner):
                return outer_n                    # bracket is an alias; key on the term
        return n

    def linked(self, a: str, b: str) -> bool:
        na, nb = normalize(a, lemma=True), normalize(b, lemma=True)
        if self.to_expansion.get(na) == nb or self.to_expansion.get(nb) == na:
            return True
        # Same reasoning as `canonical_key`: two surfaces are linked when one is
        # the other's bracketed form.
        return self.canonical_key(a) == self.canonical_key(b)