ishaq101's picture sofhiaazzhr's picture
/feat knowledge management (#20)
b68816f
Raw History Blame Contribute Delete
9.23 kB
"""Resolve references between the four artifacts. Deterministic, no LLM, no spend.
Runs after every branch has produced entries, because a link can only be made
once both of its ends exist. `GlossaryEntry` is the hub: formulas and rules
point at terms, and a term points back at the formula that defines it.
RuleEntry ──term_ids[]────► GlossaryEntry ◄──variables[].term_id── FormulaEntry
└────formula_ids[]────────────────────────────────────────────────► ▲
GlossaryEntry ──defining_formula_id──────────────────┘
Three properties this stage must keep:
1. **A dangling link is null or an empty list — never a guess.** Same rule as
span validation: an unresolvable reference is reported, not repaired. Linking
is the one stage that could quietly invent structure, so it does not.
2. **Nothing here is model-supplied.** Everything is string matching over values
already extracted and already span-checked, which is why no link can be
hallucinated.
3. **The "appears in" edge is NOT stored.** That PA occurs inside
`Production = MOHH x Qty x PA x UA x Pty` is recoverable by scanning
`formulas[].variables[]`. Do not add a field for it — a derived edge that is
also stored is an edge that can disagree with itself.
Matching is word-boundary, never substring — the same trap the ranker and the
CLI's legend stand-in both document. "PA" occurs inside "parameter", "pada",
"capacity" and "composite"; substring matching produced 126 spurious PA mentions
on a 9-page document.
"""
from __future__ import annotations
from .cluster.normalize import normalize
from .models import ClusterResult
from .rank.evidence import _word_match
def _index_terms(entries: list[dict], clustered: ClusterResult) -> dict[str, str]:
"""Normalised surface -> term_id, over canonical forms and every variant.
Variants matter: a rule naming "Physical Availability" must reach the term
stored as "PA", and only the cluster knows they are the same.
"""
by_canonical = {normalize(c.canonical): c for c in clustered.clusters}
index: dict[str, str] = {}
for entry in entries:
term_id = entry.get("term_id")
if not term_id:
continue
surfaces = [entry.get("term") or "", entry.get("full_name") or ""]
cluster = by_canonical.get(normalize(entry.get("term") or ""))
if cluster is not None:
surfaces.extend(cluster.variants)
for surface in surfaces:
key = normalize(surface)
# First writer wins: entries arrive frequency-ordered, so an
# ambiguous surface resolves to the more frequent term rather than
# to whichever happened to be processed last.
if key and key not in index:
index[key] = term_id
return index
def _resolve(text: str, index: dict[str, str]) -> list[str]:
"""Every term whose surface appears in `text`, word-boundary matched."""
haystack = normalize(text)
if not haystack:
return []
found: list[str] = []
for surface, term_id in index.items():
if term_id not in found and _word_match(surface, haystack):
found.append(term_id)
return found
def link_all(
glossary: list[dict],
rules: list[dict],
formulas: list[dict],
clustered: ClusterResult,
domain: dict | None = None,
) -> dict[str, int]:
"""Mutates the entries in place. Returns dangle counts for reporting.
The counts are returned rather than logged-and-forgotten because a link
stage that silently resolves nothing looks identical to one that works.
"""
index = _index_terms(glossary, clustered)
# formulas: variables[].symbol -> term_id
unresolved_vars = 0
for formula in formulas:
for variable in formula.get("variables") or []:
term_id = index.get(normalize(variable.get("symbol") or ""))
variable["term_id"] = term_id
if term_id is None:
unresolved_vars += 1
# glossary: term -> the formula that DEFINES it (not one it appears in)
for entry in glossary:
entry["defining_formula_id"] = _defining_formula(entry, formulas)
unresolved_defining = sum(1 for e in glossary if not e.get("defining_formula_id"))
# rules: -> terms named in the rule text, and formulas from the same chunk
formulas_by_chunk: dict[str, list[str]] = {}
for formula in formulas:
chunk_id = (formula.get("provenance") or {}).get("chunk_id")
if chunk_id:
formulas_by_chunk.setdefault(chunk_id, []).append(formula["formula_id"])
for rule in rules:
text = " ".join(
str(rule.get(f) or "") for f in ("statement", "condition", "consequence")
)
rule["term_ids"] = _resolve(text, index)
chunk_id = (rule.get("provenance") or {}).get("chunk_id")
rule["formula_ids"] = list(formulas_by_chunk.get(chunk_id or "", []))
# domain context: key_parameters[].surface -> term_id
#
# Exact-surface lookup, not `_resolve`. A key parameter IS a term name, so
# scanning it for every other term's surface would resolve "Physical
# Availability (PA)" to whichever term matched first. The rules branch wants
# "every term mentioned in this sentence"; this one wants "the term this
# name IS", and they are different questions.
unresolved_params = 0
for parameter in (domain or {}).get("key_parameters") or []:
term_id = _match_parameter(parameter.get("surface") or "", index)
parameter["term_id"] = term_id
if term_id is None:
unresolved_params += 1
return {
"terms_indexed": len(index),
"variables_unresolved": unresolved_vars,
"glossary_without_defining_formula": unresolved_defining,
"rules_without_term_link": sum(1 for r in rules if not r["term_ids"]),
"key_parameters_unresolved": unresolved_params,
}
def _match_parameter(surface: str, index: dict[str, str]) -> str | None:
"""Resolve one key-parameter surface to a term, most specific form first.
The document names its parameters as `Full Name (ABBR)` — "Physical
Availability (PA)" — while the glossary is keyed on the abbreviation with
the expansion in `full_name`. A single exact lookup therefore resolved
NOTHING on the reference document: 11 terms indexed, all 5 parameters
dangling, and five false hallucination signals sorted to the top of the
review queue, above every real term. Measured 2026-09-02.
Three candidate keys, tried longest-first so the most specific spelling
wins:
"Physical Availability (PA)" -> whole
-> "Physical Availability" (expansion)
-> "PA" (abbreviation)
Still **exact lookups**, never a scan over every indexed surface: this asks
"which term IS this name", and scanning would answer "which terms are
mentioned in it" — a different question that resolves "Physical
Availability (PA)" to whichever term happened to match first.
The abbreviation fallback is not redundant with the expansion one. On the
reference standard the document writes "Physical Availability (PA)" while
the glossary's `full_name` reads "Physical **of** Availability", so the
expansion misses and only the abbreviation connects them — the same wording
disagreement the pipeline is required to surface rather than normalise.
"""
whole = normalize(surface)
if not whole:
return None
candidates = [whole]
if "(" in surface and ")" in surface:
head, _, tail = surface.partition("(")
expansion = normalize(head)
abbrev = normalize(tail.partition(")")[0])
candidates += [c for c in (expansion, abbrev) if c]
for candidate in candidates:
term_id = index.get(candidate)
if term_id is not None:
return term_id
return None
def _defining_formula(entry: dict, formulas: list[dict]) -> str | None:
"""The formula this term is defined BY, or None.
Two signals, both deterministic: the formula's `name` names the term, or the
term sits on the left of the `=`. Appearing on the right means the term is
an *input* to that formula, which is the edge deliberately left derivable
rather than stored.
"""
surfaces = [
normalize(entry.get("term") or ""),
normalize(entry.get("full_name") or ""),
]
surfaces = [s for s in surfaces if s]
if not surfaces:
return None
for formula in formulas:
name = normalize(formula.get("name") or "")
if name and any(_word_match(s, name) for s in surfaces):
return formula["formula_id"]
latex = formula.get("formula_latex") or ""
if "=" in latex:
lhs = normalize(latex.split("=", 1)[0])
if lhs and any(_word_match(s, lhs) for s in surfaces):
return formula["formula_id"]
return None