"""Resolve references between the four artifacts. Deterministic, no LLM, no spend. Runs after every branch has produced entries, because a link can only be made once both of its ends exist. `GlossaryEntry` is the hub: formulas and rules point at terms, and a term points back at the formula that defines it. RuleEntry ──term_ids[]────► GlossaryEntry ◄──variables[].term_id── FormulaEntry └────formula_ids[]────────────────────────────────────────────────► ▲ GlossaryEntry ──defining_formula_id──────────────────┘ Three properties this stage must keep: 1. **A dangling link is null or an empty list — never a guess.** Same rule as span validation: an unresolvable reference is reported, not repaired. Linking is the one stage that could quietly invent structure, so it does not. 2. **Nothing here is model-supplied.** Everything is string matching over values already extracted and already span-checked, which is why no link can be hallucinated. 3. **The "appears in" edge is NOT stored.** That PA occurs inside `Production = MOHH x Qty x PA x UA x Pty` is recoverable by scanning `formulas[].variables[]`. Do not add a field for it — a derived edge that is also stored is an edge that can disagree with itself. Matching is word-boundary, never substring — the same trap the ranker and the CLI's legend stand-in both document. "PA" occurs inside "parameter", "pada", "capacity" and "composite"; substring matching produced 126 spurious PA mentions on a 9-page document. """ from __future__ import annotations from .cluster.normalize import normalize from .models import ClusterResult from .rank.evidence import _word_match def _index_terms(entries: list[dict], clustered: ClusterResult) -> dict[str, str]: """Normalised surface -> term_id, over canonical forms and every variant. Variants matter: a rule naming "Physical Availability" must reach the term stored as "PA", and only the cluster knows they are the same. """ by_canonical = {normalize(c.canonical): c for c in clustered.clusters} index: dict[str, str] = {} for entry in entries: term_id = entry.get("term_id") if not term_id: continue surfaces = [entry.get("term") or "", entry.get("full_name") or ""] cluster = by_canonical.get(normalize(entry.get("term") or "")) if cluster is not None: surfaces.extend(cluster.variants) for surface in surfaces: key = normalize(surface) # First writer wins: entries arrive frequency-ordered, so an # ambiguous surface resolves to the more frequent term rather than # to whichever happened to be processed last. if key and key not in index: index[key] = term_id return index def _resolve(text: str, index: dict[str, str]) -> list[str]: """Every term whose surface appears in `text`, word-boundary matched.""" haystack = normalize(text) if not haystack: return [] found: list[str] = [] for surface, term_id in index.items(): if term_id not in found and _word_match(surface, haystack): found.append(term_id) return found def link_all( glossary: list[dict], rules: list[dict], formulas: list[dict], clustered: ClusterResult, domain: dict | None = None, ) -> dict[str, int]: """Mutates the entries in place. Returns dangle counts for reporting. The counts are returned rather than logged-and-forgotten because a link stage that silently resolves nothing looks identical to one that works. """ index = _index_terms(glossary, clustered) # formulas: variables[].symbol -> term_id unresolved_vars = 0 for formula in formulas: for variable in formula.get("variables") or []: term_id = index.get(normalize(variable.get("symbol") or "")) variable["term_id"] = term_id if term_id is None: unresolved_vars += 1 # glossary: term -> the formula that DEFINES it (not one it appears in) for entry in glossary: entry["defining_formula_id"] = _defining_formula(entry, formulas) unresolved_defining = sum(1 for e in glossary if not e.get("defining_formula_id")) # rules: -> terms named in the rule text, and formulas from the same chunk formulas_by_chunk: dict[str, list[str]] = {} for formula in formulas: chunk_id = (formula.get("provenance") or {}).get("chunk_id") if chunk_id: formulas_by_chunk.setdefault(chunk_id, []).append(formula["formula_id"]) for rule in rules: text = " ".join( str(rule.get(f) or "") for f in ("statement", "condition", "consequence") ) rule["term_ids"] = _resolve(text, index) chunk_id = (rule.get("provenance") or {}).get("chunk_id") rule["formula_ids"] = list(formulas_by_chunk.get(chunk_id or "", [])) # domain context: key_parameters[].surface -> term_id # # Exact-surface lookup, not `_resolve`. A key parameter IS a term name, so # scanning it for every other term's surface would resolve "Physical # Availability (PA)" to whichever term matched first. The rules branch wants # "every term mentioned in this sentence"; this one wants "the term this # name IS", and they are different questions. unresolved_params = 0 for parameter in (domain or {}).get("key_parameters") or []: term_id = _match_parameter(parameter.get("surface") or "", index) parameter["term_id"] = term_id if term_id is None: unresolved_params += 1 return { "terms_indexed": len(index), "variables_unresolved": unresolved_vars, "glossary_without_defining_formula": unresolved_defining, "rules_without_term_link": sum(1 for r in rules if not r["term_ids"]), "key_parameters_unresolved": unresolved_params, } def _match_parameter(surface: str, index: dict[str, str]) -> str | None: """Resolve one key-parameter surface to a term, most specific form first. The document names its parameters as `Full Name (ABBR)` — "Physical Availability (PA)" — while the glossary is keyed on the abbreviation with the expansion in `full_name`. A single exact lookup therefore resolved NOTHING on the reference document: 11 terms indexed, all 5 parameters dangling, and five false hallucination signals sorted to the top of the review queue, above every real term. Measured 2026-09-02. Three candidate keys, tried longest-first so the most specific spelling wins: "Physical Availability (PA)" -> whole -> "Physical Availability" (expansion) -> "PA" (abbreviation) Still **exact lookups**, never a scan over every indexed surface: this asks "which term IS this name", and scanning would answer "which terms are mentioned in it" — a different question that resolves "Physical Availability (PA)" to whichever term happened to match first. The abbreviation fallback is not redundant with the expansion one. On the reference standard the document writes "Physical Availability (PA)" while the glossary's `full_name` reads "Physical **of** Availability", so the expansion misses and only the abbreviation connects them — the same wording disagreement the pipeline is required to surface rather than normalise. """ whole = normalize(surface) if not whole: return None candidates = [whole] if "(" in surface and ")" in surface: head, _, tail = surface.partition("(") expansion = normalize(head) abbrev = normalize(tail.partition(")")[0]) candidates += [c for c in (expansion, abbrev) if c] for candidate in candidates: term_id = index.get(candidate) if term_id is not None: return term_id return None def _defining_formula(entry: dict, formulas: list[dict]) -> str | None: """The formula this term is defined BY, or None. Two signals, both deterministic: the formula's `name` names the term, or the term sits on the left of the `=`. Appearing on the right means the term is an *input* to that formula, which is the edge deliberately left derivable rather than stored. """ surfaces = [ normalize(entry.get("term") or ""), normalize(entry.get("full_name") or ""), ] surfaces = [s for s in surfaces if s] if not surfaces: return None for formula in formulas: name = normalize(formula.get("name") or "") if name and any(_word_match(s, name) for s in surfaces): return formula["formula_id"] latex = formula.get("formula_latex") or "" if "=" in latex: lhs = normalize(latex.split("=", 1)[0]) if lhs and any(_word_match(s, lhs) for s in surfaces): return formula["formula_id"] return None