File size: 9,228 Bytes
b68816f | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 | """Resolve references between the four artifacts. Deterministic, no LLM, no spend.
Runs after every branch has produced entries, because a link can only be made
once both of its ends exist. `GlossaryEntry` is the hub: formulas and rules
point at terms, and a term points back at the formula that defines it.
RuleEntry ββterm_ids[]βββββΊ GlossaryEntry βββvariables[].term_idββ FormulaEntry
βββββformula_ids[]βββββββββββββββββββββββββββββββββββββββββββββββββΊ β²
GlossaryEntry ββdefining_formula_idβββββββββββββββββββ
Three properties this stage must keep:
1. **A dangling link is null or an empty list β never a guess.** Same rule as
span validation: an unresolvable reference is reported, not repaired. Linking
is the one stage that could quietly invent structure, so it does not.
2. **Nothing here is model-supplied.** Everything is string matching over values
already extracted and already span-checked, which is why no link can be
hallucinated.
3. **The "appears in" edge is NOT stored.** That PA occurs inside
`Production = MOHH x Qty x PA x UA x Pty` is recoverable by scanning
`formulas[].variables[]`. Do not add a field for it β a derived edge that is
also stored is an edge that can disagree with itself.
Matching is word-boundary, never substring β the same trap the ranker and the
CLI's legend stand-in both document. "PA" occurs inside "parameter", "pada",
"capacity" and "composite"; substring matching produced 126 spurious PA mentions
on a 9-page document.
"""
from __future__ import annotations
from .cluster.normalize import normalize
from .models import ClusterResult
from .rank.evidence import _word_match
def _index_terms(entries: list[dict], clustered: ClusterResult) -> dict[str, str]:
"""Normalised surface -> term_id, over canonical forms and every variant.
Variants matter: a rule naming "Physical Availability" must reach the term
stored as "PA", and only the cluster knows they are the same.
"""
by_canonical = {normalize(c.canonical): c for c in clustered.clusters}
index: dict[str, str] = {}
for entry in entries:
term_id = entry.get("term_id")
if not term_id:
continue
surfaces = [entry.get("term") or "", entry.get("full_name") or ""]
cluster = by_canonical.get(normalize(entry.get("term") or ""))
if cluster is not None:
surfaces.extend(cluster.variants)
for surface in surfaces:
key = normalize(surface)
# First writer wins: entries arrive frequency-ordered, so an
# ambiguous surface resolves to the more frequent term rather than
# to whichever happened to be processed last.
if key and key not in index:
index[key] = term_id
return index
def _resolve(text: str, index: dict[str, str]) -> list[str]:
"""Every term whose surface appears in `text`, word-boundary matched."""
haystack = normalize(text)
if not haystack:
return []
found: list[str] = []
for surface, term_id in index.items():
if term_id not in found and _word_match(surface, haystack):
found.append(term_id)
return found
def link_all(
glossary: list[dict],
rules: list[dict],
formulas: list[dict],
clustered: ClusterResult,
domain: dict | None = None,
) -> dict[str, int]:
"""Mutates the entries in place. Returns dangle counts for reporting.
The counts are returned rather than logged-and-forgotten because a link
stage that silently resolves nothing looks identical to one that works.
"""
index = _index_terms(glossary, clustered)
# formulas: variables[].symbol -> term_id
unresolved_vars = 0
for formula in formulas:
for variable in formula.get("variables") or []:
term_id = index.get(normalize(variable.get("symbol") or ""))
variable["term_id"] = term_id
if term_id is None:
unresolved_vars += 1
# glossary: term -> the formula that DEFINES it (not one it appears in)
for entry in glossary:
entry["defining_formula_id"] = _defining_formula(entry, formulas)
unresolved_defining = sum(1 for e in glossary if not e.get("defining_formula_id"))
# rules: -> terms named in the rule text, and formulas from the same chunk
formulas_by_chunk: dict[str, list[str]] = {}
for formula in formulas:
chunk_id = (formula.get("provenance") or {}).get("chunk_id")
if chunk_id:
formulas_by_chunk.setdefault(chunk_id, []).append(formula["formula_id"])
for rule in rules:
text = " ".join(
str(rule.get(f) or "") for f in ("statement", "condition", "consequence")
)
rule["term_ids"] = _resolve(text, index)
chunk_id = (rule.get("provenance") or {}).get("chunk_id")
rule["formula_ids"] = list(formulas_by_chunk.get(chunk_id or "", []))
# domain context: key_parameters[].surface -> term_id
#
# Exact-surface lookup, not `_resolve`. A key parameter IS a term name, so
# scanning it for every other term's surface would resolve "Physical
# Availability (PA)" to whichever term matched first. The rules branch wants
# "every term mentioned in this sentence"; this one wants "the term this
# name IS", and they are different questions.
unresolved_params = 0
for parameter in (domain or {}).get("key_parameters") or []:
term_id = _match_parameter(parameter.get("surface") or "", index)
parameter["term_id"] = term_id
if term_id is None:
unresolved_params += 1
return {
"terms_indexed": len(index),
"variables_unresolved": unresolved_vars,
"glossary_without_defining_formula": unresolved_defining,
"rules_without_term_link": sum(1 for r in rules if not r["term_ids"]),
"key_parameters_unresolved": unresolved_params,
}
def _match_parameter(surface: str, index: dict[str, str]) -> str | None:
"""Resolve one key-parameter surface to a term, most specific form first.
The document names its parameters as `Full Name (ABBR)` β "Physical
Availability (PA)" β while the glossary is keyed on the abbreviation with
the expansion in `full_name`. A single exact lookup therefore resolved
NOTHING on the reference document: 11 terms indexed, all 5 parameters
dangling, and five false hallucination signals sorted to the top of the
review queue, above every real term. Measured 2026-09-02.
Three candidate keys, tried longest-first so the most specific spelling
wins:
"Physical Availability (PA)" -> whole
-> "Physical Availability" (expansion)
-> "PA" (abbreviation)
Still **exact lookups**, never a scan over every indexed surface: this asks
"which term IS this name", and scanning would answer "which terms are
mentioned in it" β a different question that resolves "Physical
Availability (PA)" to whichever term happened to match first.
The abbreviation fallback is not redundant with the expansion one. On the
reference standard the document writes "Physical Availability (PA)" while
the glossary's `full_name` reads "Physical **of** Availability", so the
expansion misses and only the abbreviation connects them β the same wording
disagreement the pipeline is required to surface rather than normalise.
"""
whole = normalize(surface)
if not whole:
return None
candidates = [whole]
if "(" in surface and ")" in surface:
head, _, tail = surface.partition("(")
expansion = normalize(head)
abbrev = normalize(tail.partition(")")[0])
candidates += [c for c in (expansion, abbrev) if c]
for candidate in candidates:
term_id = index.get(candidate)
if term_id is not None:
return term_id
return None
def _defining_formula(entry: dict, formulas: list[dict]) -> str | None:
"""The formula this term is defined BY, or None.
Two signals, both deterministic: the formula's `name` names the term, or the
term sits on the left of the `=`. Appearing on the right means the term is
an *input* to that formula, which is the edge deliberately left derivable
rather than stored.
"""
surfaces = [
normalize(entry.get("term") or ""),
normalize(entry.get("full_name") or ""),
]
surfaces = [s for s in surfaces if s]
if not surfaces:
return None
for formula in formulas:
name = normalize(formula.get("name") or "")
if name and any(_word_match(s, name) for s in surfaces):
return formula["formula_id"]
latex = formula.get("formula_latex") or ""
if "=" in latex:
lhs = normalize(latex.split("=", 1)[0])
if lhs and any(_word_match(s, lhs) for s in surfaces):
return formula["formula_id"]
return None
|