File size: 9,228 Bytes
b68816f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
"""Resolve references between the four artifacts. Deterministic, no LLM, no spend.

Runs after every branch has produced entries, because a link can only be made
once both of its ends exist. `GlossaryEntry` is the hub: formulas and rules
point at terms, and a term points back at the formula that defines it.

    RuleEntry ──term_ids[]────► GlossaryEntry ◄──variables[].term_id── FormulaEntry
         └────formula_ids[]────────────────────────────────────────────────► β–²
                        GlossaryEntry ──defining_formula_idβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜

Three properties this stage must keep:

1. **A dangling link is null or an empty list β€” never a guess.** Same rule as
   span validation: an unresolvable reference is reported, not repaired. Linking
   is the one stage that could quietly invent structure, so it does not.
2. **Nothing here is model-supplied.** Everything is string matching over values
   already extracted and already span-checked, which is why no link can be
   hallucinated.
3. **The "appears in" edge is NOT stored.** That PA occurs inside
   `Production = MOHH x Qty x PA x UA x Pty` is recoverable by scanning
   `formulas[].variables[]`. Do not add a field for it β€” a derived edge that is
   also stored is an edge that can disagree with itself.

Matching is word-boundary, never substring β€” the same trap the ranker and the
CLI's legend stand-in both document. "PA" occurs inside "parameter", "pada",
"capacity" and "composite"; substring matching produced 126 spurious PA mentions
on a 9-page document.
"""

from __future__ import annotations

from .cluster.normalize import normalize
from .models import ClusterResult
from .rank.evidence import _word_match


def _index_terms(entries: list[dict], clustered: ClusterResult) -> dict[str, str]:
    """Normalised surface -> term_id, over canonical forms and every variant.

    Variants matter: a rule naming "Physical Availability" must reach the term
    stored as "PA", and only the cluster knows they are the same.
    """
    by_canonical = {normalize(c.canonical): c for c in clustered.clusters}
    index: dict[str, str] = {}
    for entry in entries:
        term_id = entry.get("term_id")
        if not term_id:
            continue
        surfaces = [entry.get("term") or "", entry.get("full_name") or ""]
        cluster = by_canonical.get(normalize(entry.get("term") or ""))
        if cluster is not None:
            surfaces.extend(cluster.variants)
        for surface in surfaces:
            key = normalize(surface)
            # First writer wins: entries arrive frequency-ordered, so an
            # ambiguous surface resolves to the more frequent term rather than
            # to whichever happened to be processed last.
            if key and key not in index:
                index[key] = term_id
    return index


def _resolve(text: str, index: dict[str, str]) -> list[str]:
    """Every term whose surface appears in `text`, word-boundary matched."""
    haystack = normalize(text)
    if not haystack:
        return []
    found: list[str] = []
    for surface, term_id in index.items():
        if term_id not in found and _word_match(surface, haystack):
            found.append(term_id)
    return found


def link_all(
    glossary: list[dict],
    rules: list[dict],
    formulas: list[dict],
    clustered: ClusterResult,
    domain: dict | None = None,
) -> dict[str, int]:
    """Mutates the entries in place. Returns dangle counts for reporting.

    The counts are returned rather than logged-and-forgotten because a link
    stage that silently resolves nothing looks identical to one that works.
    """
    index = _index_terms(glossary, clustered)

    # formulas: variables[].symbol -> term_id
    unresolved_vars = 0
    for formula in formulas:
        for variable in formula.get("variables") or []:
            term_id = index.get(normalize(variable.get("symbol") or ""))
            variable["term_id"] = term_id
            if term_id is None:
                unresolved_vars += 1

    # glossary: term -> the formula that DEFINES it (not one it appears in)
    for entry in glossary:
        entry["defining_formula_id"] = _defining_formula(entry, formulas)
    unresolved_defining = sum(1 for e in glossary if not e.get("defining_formula_id"))

    # rules: -> terms named in the rule text, and formulas from the same chunk
    formulas_by_chunk: dict[str, list[str]] = {}
    for formula in formulas:
        chunk_id = (formula.get("provenance") or {}).get("chunk_id")
        if chunk_id:
            formulas_by_chunk.setdefault(chunk_id, []).append(formula["formula_id"])

    for rule in rules:
        text = " ".join(
            str(rule.get(f) or "") for f in ("statement", "condition", "consequence")
        )
        rule["term_ids"] = _resolve(text, index)
        chunk_id = (rule.get("provenance") or {}).get("chunk_id")
        rule["formula_ids"] = list(formulas_by_chunk.get(chunk_id or "", []))

    # domain context: key_parameters[].surface -> term_id
    #
    # Exact-surface lookup, not `_resolve`. A key parameter IS a term name, so
    # scanning it for every other term's surface would resolve "Physical
    # Availability (PA)" to whichever term matched first. The rules branch wants
    # "every term mentioned in this sentence"; this one wants "the term this
    # name IS", and they are different questions.
    unresolved_params = 0
    for parameter in (domain or {}).get("key_parameters") or []:
        term_id = _match_parameter(parameter.get("surface") or "", index)
        parameter["term_id"] = term_id
        if term_id is None:
            unresolved_params += 1

    return {
        "terms_indexed": len(index),
        "variables_unresolved": unresolved_vars,
        "glossary_without_defining_formula": unresolved_defining,
        "rules_without_term_link": sum(1 for r in rules if not r["term_ids"]),
        "key_parameters_unresolved": unresolved_params,
    }


def _match_parameter(surface: str, index: dict[str, str]) -> str | None:
    """Resolve one key-parameter surface to a term, most specific form first.

    The document names its parameters as `Full Name (ABBR)` β€” "Physical
    Availability (PA)" β€” while the glossary is keyed on the abbreviation with
    the expansion in `full_name`. A single exact lookup therefore resolved
    NOTHING on the reference document: 11 terms indexed, all 5 parameters
    dangling, and five false hallucination signals sorted to the top of the
    review queue, above every real term. Measured 2026-09-02.

    Three candidate keys, tried longest-first so the most specific spelling
    wins:

        "Physical Availability (PA)" -> whole
                                     -> "Physical Availability"   (expansion)
                                     -> "PA"                      (abbreviation)

    Still **exact lookups**, never a scan over every indexed surface: this asks
    "which term IS this name", and scanning would answer "which terms are
    mentioned in it" β€” a different question that resolves "Physical
    Availability (PA)" to whichever term happened to match first.

    The abbreviation fallback is not redundant with the expansion one. On the
    reference standard the document writes "Physical Availability (PA)" while
    the glossary's `full_name` reads "Physical **of** Availability", so the
    expansion misses and only the abbreviation connects them β€” the same wording
    disagreement the pipeline is required to surface rather than normalise.
    """
    whole = normalize(surface)
    if not whole:
        return None

    candidates = [whole]
    if "(" in surface and ")" in surface:
        head, _, tail = surface.partition("(")
        expansion = normalize(head)
        abbrev = normalize(tail.partition(")")[0])
        candidates += [c for c in (expansion, abbrev) if c]

    for candidate in candidates:
        term_id = index.get(candidate)
        if term_id is not None:
            return term_id
    return None


def _defining_formula(entry: dict, formulas: list[dict]) -> str | None:
    """The formula this term is defined BY, or None.

    Two signals, both deterministic: the formula's `name` names the term, or the
    term sits on the left of the `=`. Appearing on the right means the term is
    an *input* to that formula, which is the edge deliberately left derivable
    rather than stored.
    """
    surfaces = [
        normalize(entry.get("term") or ""),
        normalize(entry.get("full_name") or ""),
    ]
    surfaces = [s for s in surfaces if s]
    if not surfaces:
        return None

    for formula in formulas:
        name = normalize(formula.get("name") or "")
        if name and any(_word_match(s, name) for s in surfaces):
            return formula["formula_id"]

        latex = formula.get("formula_latex") or ""
        if "=" in latex:
            lhs = normalize(latex.split("=", 1)[0])
            if lhs and any(_word_match(s, lhs) for s in surfaces):
                return formula["formula_id"]
    return None